From 1017ed7902359f90ee6fc5aa4bfcf9982c338c74 Mon Sep 17 00:00:00 2001 From: steveya Date: Sat, 12 Sep 2026 23:24:15 -0400 Subject: [PATCH 1/3] Author E22 cap-strip proof contract and exercised controls --- LIMITATIONS.md | 3 +- TASKS_PROOF_LEGACY.yaml | 147 ++++++- TASKS_PROOF_LEGACY_BASELINE.yaml | 6 +- doc/plan/active__legacy-task-migration-map.md | 2 +- doc/plan/active__task-manifest-integrity.md | 2 +- .../implementation_journey_prompt_to_price.md | 16 + docs/quant/pricing_stack.rst | 21 + docs/user_guide/pricing.rst | 10 + tests/test_tasks/test_e22_cap_strip.py | 410 ++++++++++++++++++ trellis/agent/benchmark_contracts.py | 16 + trellis/agent/executor.py | 43 +- trellis/agent/planner.py | 2 + trellis/agent/semantic_contracts.py | 5 +- trellis/agent/task_manifest_validation.py | 135 ++++++ trellis/agent/task_runtime.py | 14 + 15 files changed, 812 insertions(+), 20 deletions(-) create mode 100644 tests/test_tasks/test_e22_cap_strip.py diff --git a/LIMITATIONS.md b/LIMITATIONS.md index d3a80f2e..a61b3e7f 100644 --- a/LIMITATIONS.md +++ b/LIMITATIONS.md @@ -69,7 +69,8 @@ ground truth until revalidated. |---|-----------|--------|-------| | L49 | **Agent-cycle and institutional approval evidence are governed internal review surfaces, not external model certification** — Trellis now exposes a stable `agent_cycle` result surface for quant/critic/arbiter/model-validator evidence, model-promotion eligibility, and benchmark trigger rates, and `policy_bundle.production.institutional` can require approval, model-review, snapshot, run-artifact, and audit-bundle evidence before production execution, but these surfaces only certify recorded internal governance evidence for the run or model version | Product and desk review surfaces can show why a cycle passed, failed, was unavailable, or lacked required institutional approval artifacts, but they must not be read as external model approval, regulatory sign-off, xVA/FpML coverage, or correctness beyond the recorded validation scope | `trellis/agent/cycle_surface.py`, `trellis/agent/task_runtime.py`, `trellis/platform/models.py`, `trellis/platform/policies.py`, `trellis/platform/services/pricing_service.py`, `docs/developer/audit_and_observability.rst`, `docs/developer/hosting_and_configuration.rst`, `docs/user_guide/pricing.rst` | | L57 | **Typed comparison-target coherence exposes rather than fills missing numerical composition** — Explicit targets carry canonical method, route/binding, variant, validation, and semantic identities, and the runtime rejects partial explicit target sets, ambiguous references, missing declarations, and unbound shared artifacts before pricing. Explicit semantic axes are now projected onto the per-target `ProductIR` before route selection, which prevents serialized target prose from reclassifying terminal spread baskets and lets T102/T126 bind their independent Stulz, Kirk, Monte Carlo, and Hurd-Zhou lanes. T102 now proves its Monte Carlo variant through authored `n_paths`, `n_steps`, `seed`, and `mc_method` spec overrides while its Stulz reference binds the raw analytical kernel. Other variant execution still must be proven by spec overrides or a canonical full-contract executable declaration; sparse legacy targets still rely on visibly inferred contracts, and `T13` remains an exact-bucket semantic-guard canary whose cached analytical artifact cannot honestly represent both theta-PDE variants and the analytical reference | Coherence can select and prove reusable numerical composition when it exists, but it deliberately does not invent missing methods, infer undeclared semantics for sparse legacy targets, or let one cached artifact impersonate several variants | `trellis/agent/comparison_target_contracts.py`, `trellis/agent/assembly_tools.py`, `trellis/agent/task_runtime.py`, `trellis/agent/executor.py`, `TASKS_PROOF_LEGACY.yaml`, `docs/developer/task_and_eval_loops.rst`, `docs/quant/pricing_stack.rst` | -| L64 | **Legacy proof-task contracts remain broadly incomplete** — the task-manifest gate now validates all modern corpus envelopes strictly and freezes the legacy corpus's exact field-level debt and normalized task content behind a checked baseline. After the authored T02, T17, and T102 repairs, that baseline contains 590 exact issue identities across 122 incomplete retained rows; those rows still lack one or more authored descriptions, economic contracts, market contracts, acceptance criteria, or explicit execution/hold dispositions. Product-specific field sufficiency remains owned by the semantic validators and repair tickets | New or worsened structural manifest debt fails the corpus gate, and the main task runner plus specific-id rerunner reject selected incomplete legacy rows before default market construction or code generation. The exact authored T02, T17, and T102 rows pass that boundary without weakening the remaining debt. The legacy baseline is only a migration guard; it must not be interpreted as evidence that title-only rows are priceable, that every specialized proof harness is governed by the main runner boundary, or that every structurally valid modern product contract is semantically complete | `TASKS_PROOF_LEGACY.yaml`, `TASKS_PROOF_LEGACY_BASELINE.yaml`, `trellis/agent/task_manifest_validation.py`, `scripts/validate_task_manifests.py`, `scripts/run_tasks.py`, `scripts/rerun_ids.py`, `doc/plan/active__task-manifest-integrity.md` | +| L64 | **Legacy proof-task contracts remain broadly incomplete** — the task-manifest gate now validates all modern corpus envelopes strictly and freezes the legacy corpus's exact field-level debt and normalized task content behind a checked baseline. After the authored T02, T17, T102, and E22 repairs, that baseline contains 586 exact issue identities across 121 incomplete retained rows; those rows still lack one or more authored descriptions, economic contracts, market contracts, acceptance criteria, or explicit execution/hold dispositions. Product-specific field sufficiency remains owned by the semantic validators and repair tickets | New or worsened structural manifest debt fails the corpus gate, and the main task runner plus specific-id rerunner reject selected incomplete legacy rows before default market construction or code generation. The exact authored T02, T17, T102, and E22 rows pass that boundary without weakening the remaining debt. The legacy baseline is only a migration guard; it must not be interpreted as evidence that title-only rows are priceable, that every specialized proof harness is governed by the main runner boundary, or that every structurally valid modern product contract is semantically complete | `TASKS_PROOF_LEGACY.yaml`, `TASKS_PROOF_LEGACY_BASELINE.yaml`, `trellis/agent/task_manifest_validation.py`, `scripts/validate_task_manifests.py`, `scripts/run_tasks.py`, `scripts/rerun_ids.py`, `doc/plan/active__task-manifest-integrity.md` | +| L68 | **E22 cap-strip Monte Carlo is a forward-marginal proof** — the authored USD cap has twenty quarterly unadjusted periods, explicit fixing and payment dates, ACT/360 accrual, ACT/365 option time, ACT/ACT ISDA curve time, named OIS/SOFR-3M/Black-vol inputs, and seeded antithetic sampling. Both lanes value the same sum of individual caplet expectations | The 0.5% analytical-reference comparison supports this bounded proof. The Monte Carlo helper does not simulate a short-rate path or a joint forward-rate process, and its legacy n_steps, mean_reversion and sigma arguments are ignored; E22 cannot declare those controls. The proof does not establish calibration, shifted/normal volatility, seasoned fixing or production holiday support | `TASKS_PROOF_LEGACY.yaml`, `trellis/models/rate_cap_floor.py`, `trellis/agent/task_manifest_validation.py`, `trellis/agent/executor.py`, `tests/test_tasks/test_e22_cap_strip.py`, `docs/quant/pricing_stack.rst` | | L65 | **Callable-bond coupons are scalar fixed-rate only** — the checked callable-bond cashflow, compiler, lattice, and PDE routes accept one scalar coupon rate rather than a dated variable-coupon schedule. Legacy task T09 asks for a step-up callable bond but does not author the coupon rates or effective dates, so its validated task contract now fails closed with an exact `variable_coupon_schedule` blocker and zero build attempts instead of using the old title-derived flat 5% fixture | Trellis can price the bounded fixed-coupon callable-bond cohort, but it cannot reasonably price step-up, step-down, floating, or otherwise variable-coupon callable bonds. T09 remains an expected honest block until QUA-1251 adds a reusable dated coupon primitive and an explicit schedule | `trellis/models/short_rate_fixed_income.py`, `trellis/instruments/callable_bond.py`, `trellis/models/callable_bond_pde.py`, `trellis/execution/compiler.py`, `trellis/agent/task_runtime.py`, `TASKS_PROOF_LEGACY.yaml`, `tests/test_agent/test_task_runtime.py` | | L66 | **Physical Bermudan swaption lattice support is a bounded one-factor, static-basis composition** — the strict route preserves explicit co-terminal swap tails, separate named discount/forecast curves, complete supported leg conventions, a provenance-complete named constant-parameter Hull-White set, and authored uniform-grid controls. It supports physical settlement, simple floating coupons with a deterministic additive forward basis, and ACT/365F model time. Stochastic basis, reset/payment convexity, compounded overnight coupons, amortizing or scheduled notionals, seasoned or pre-started fixed tails requiring accrued-settlement treatment, ACT/ACT ICMA coupon accrual without explicit quasi-coupon reference periods, parameterized day-of-month rolls, non-shipped calendar aliases, cash/annuity settlement, term-structured model volatility, multi-factor rates, native Greeks, and production convergence/error governance remain unsupported | Trellis can reasonably price the checked bounded physical dual-curve contract only when each adjusted first fixed accrual start is on or after exercise; it must fail closed rather than value a whole already-started fixed coupon, reinterpret richer Bermudan swaptions through the legacy T04 helper, use a European/Black fallback, or approximate a convention mapping. P005 is the exact executable evidence: its strict lattice lane prices and remains visible even though the paired Monte Carlo lane honestly blocks | `trellis/models/rate_swap_tail.py`, `trellis/models/hull_white_parameters.py`, `trellis/agent/semantic_contracts.py`, `trellis/agent/executor.py`, `trellis/agent/knowledge/canonical/routes.yaml`, `TASKS_EXTENSION.yaml`, `tests/test_models/test_rate_swap_tail.py`, `tests/test_agent/test_physical_bermudan_swaption_semantics.py`, `tests/test_tasks/test_p005_physical_bermudan_swaption.py`, `docs/quant/lattice_algebra.rst`, `docs/user_guide/pricing.rst` | | L58 | **FpML normalization is bounded to fixed-float IRS, physical European swaption, and scheduled cap/floor cohorts** — Support-contract version 1.0.0 distinguishes secure inspection, economic normalization, executable structural lowering, and paired conformance. `make_fpml_request(...)` and `trellis.io.fpml` securely inspect inline UTF-8 FpML 5.13 confirmation `dataDocument` payloads and normalize one regular, single-currency, constant-notional fixed-float swap into `StaticLegContractIR`, one physically settled European payer/receiver swaption into `ContractIR` with the complete swap nested under `underlying_contract`, or one regular single-currency constant-strike cap/floor into the existing signed `PeriodRateOptionStripLeg`. Swaptions reuse the structural resolved Black-76 declaration; cap/floors reuse the existing static strip declaration; historical settled premiums are reported separately and excluded from contract identity. `TASKS_FPML_CONFORMANCE.yaml` pairs all three admitted cohorts with independently specified native contracts and proves identity, projection, structural selection, market binding, price, and non-economic envelope invariance; its negative cohort certifies exact honest blockers with zero agent calls. This evidence does not widen support. Trellis still does not perform complete XSD validation, support other views/versions, resolve external references, bind imported seasoned coupons to historical fixing histories, classify vendor extension children, or normalize amortizing, compounding, stubbed, end-of-month or clamped high-day, cross-currency, OIS, inflation, lifecycle, package, cash-settled/Bermudan/American/partial/automatic/straddle swaption, unsettled-premium, cap/floor collar, stepped-strike, averaged, geared/spread, early-terminable, or other product forms | An admitted swap, physical European swaption, or scheduled cap/floor strip can price deterministically through shared structural execution when the caller declares a valuation party and valuation date and all cohort constraints hold. Unclassified extension children and all other FpML economics remain fail-closed with exact import, clarification, conflict, or unsupported-feature blockers; this is not general FpML pricing coverage | `TASKS_FPML_CONFORMANCE.yaml`, `trellis/io/fpml/`, `trellis/agent/fpml_conformance.py`, `trellis/agent/contract_ir.py`, `trellis/agent/static_leg_contract.py`, `trellis/agent/imported_documents.py`, `trellis/agent/platform_requests.py`, `trellis/platform/executor.py`, `docs/developer/fpml_support_matrix.rst`, `docs/developer/fpml_import.rst`, `docs/quant/contract_ir.rst`, `docs/quant/static_leg_contract_ir.rst`, `doc/plan/draft__fpml-interoperability-roadmap.md` | diff --git a/TASKS_PROOF_LEGACY.yaml b/TASKS_PROOF_LEGACY.yaml index f71c88b5..39e97d5b 100644 --- a/TASKS_PROOF_LEGACY.yaml +++ b/TASKS_PROOF_LEGACY.yaml @@ -2400,17 +2400,112 @@ tasks: analytical: black_scholes status: pending - id: E22 - title: 'Cap/floor: Black caplet stack vs MC rate simulation' + title: 'Cap strip: Black caplets vs sampled forward marginals' + description: >- + Price the authored USD cap strip under the named rates scenario. Compare + discounted Black-76 caplets with seeded antithetic Monte Carlo of independent + lognormal caplet forward marginals, reporting holder present value in USD. + task_disposition: executable_pricing + instrument_type: period_rate_option_strip + market_scenario_id: usd_rates_smile + validation_policy: invariants_and_cross_method construct: - analytical - monte_carlo new_component: null - market: - source: mock - as_of: 2024-11-15 - discount_curve: usd_ois - forecast_curve: USD-SOFR-3M - vol_surface: usd_rates_smile + benchmark_contract: + product: period_rate_option_strip + cap_floor: cap + currency: USD + notional: 1000000.0 + strike: 0.04 + valuation_date: '2024-11-15' + start_date: '2025-02-15' + end_date: '2030-02-15' + payment_frequency: quarterly + day_count: ACT/360 + model_time_day_count: ACT/365 + discount_curve_day_count: ACT/ACT ISDA + forecast_curve_day_count: ACT/ACT ISDA + calendar_name: weekend_only + business_day_adjustment: unadjusted + fixing_rule: accrual_start + payment_rule: accrual_end + fixing_lag_days: 0 + payment_lag_days: 0 + accrual_dates: + - '2025-02-15' + - '2025-05-15' + - '2025-08-15' + - '2025-11-15' + - '2026-02-15' + - '2026-05-15' + - '2026-08-15' + - '2026-11-15' + - '2027-02-15' + - '2027-05-15' + - '2027-08-15' + - '2027-11-15' + - '2028-02-15' + - '2028-05-15' + - '2028-08-15' + - '2028-11-15' + - '2029-02-15' + - '2029-05-15' + - '2029-08-15' + - '2029-11-15' + - '2030-02-15' + fixing_dates: + - '2025-02-15' + - '2025-05-15' + - '2025-08-15' + - '2025-11-15' + - '2026-02-15' + - '2026-05-15' + - '2026-08-15' + - '2026-11-15' + - '2027-02-15' + - '2027-05-15' + - '2027-08-15' + - '2027-11-15' + - '2028-02-15' + - '2028-05-15' + - '2028-08-15' + - '2028-11-15' + - '2029-02-15' + - '2029-05-15' + - '2029-08-15' + - '2029-11-15' + payment_dates: + - '2025-05-15' + - '2025-08-15' + - '2025-11-15' + - '2026-02-15' + - '2026-05-15' + - '2026-08-15' + - '2026-11-15' + - '2027-02-15' + - '2027-05-15' + - '2027-08-15' + - '2027-11-15' + - '2028-02-15' + - '2028-05-15' + - '2028-08-15' + - '2028-11-15' + - '2029-02-15' + - '2029-05-15' + - '2029-08-15' + - '2029-11-15' + - '2030-02-15' + rate_index: USD-SOFR-3M + model: black + mc_distribution: independent_lognormal_forward_marginals + sampling: antithetic + n_paths: 100000 + seed: 42 + valuation_measure: holder_present_value + output_unit: currency_amount + output_currency: USD market_assertions: requires: - discount_curve @@ -2422,8 +2517,42 @@ tasks: vol_surface: usd_rates_smile cross_validate: internal: - - mc_rate_cap - analytical: black76_cap + - analytical + - monte_carlo + reference_target: analytical + relations: + monte_carlo: within_tolerance + tolerance_pct: 0.5 + tolerance_unit: percent_of_reference_price + output_unit: currency_amount + output_currency: USD + target_contracts: + analytical: + method: analytical + route_family: analytical + backend_binding_id: trellis.models.rate_cap_floor.price_rate_cap_floor_strip_analytical + validation_bundle_id: analytical:cap + payoff_family: period_rate_option_strip + exercise_style: none + model_family: interest_rate + observation_style: fixed_schedule + variant_parameters: + model: black + monte_carlo: + method: monte_carlo + route_family: monte_carlo + backend_binding_id: trellis.models.rate_cap_floor.price_rate_cap_floor_strip_monte_carlo + validation_bundle_id: monte_carlo:cap + payoff_family: period_rate_option_strip + exercise_style: none + model_family: interest_rate + observation_style: fixed_schedule + variant_parameters: + distribution: independent_lognormal_forward_marginals + sampling: antithetic + spec_overrides: + n_paths: 100000 + seed: 42 status: pending - id: E23 title: 'European equity call under local vol: PDE vs MC' diff --git a/TASKS_PROOF_LEGACY_BASELINE.yaml b/TASKS_PROOF_LEGACY_BASELINE.yaml index df2021c6..d3ff7804 100644 --- a/TASKS_PROOF_LEGACY_BASELINE.yaml +++ b/TASKS_PROOF_LEGACY_BASELINE.yaml @@ -1,8 +1,8 @@ version: 1 manifest: TASKS_PROOF_LEGACY.yaml -issue_count: 590 -issue_digest: 454a9b098b6d8de5754b9e151004670d2b33c667155ca8eea2bc928b5705acd6 -task_fingerprint: 55b9bcdd3364c0e1c0ca6d96f3cc9caf17e7cf4a78a40af92541804e0c368876 +issue_count: 586 +issue_digest: fe69e77b4f3804741c12d59f384608ec4fd5861ce507640902e763b46e6b7f4c +task_fingerprint: 5c556d9a0f00b65ffbcc405aef606c35ab95f0a0c3ad243035d31d2542e2ffaf policy: exact_issue_identity_and_task_content note: >- This baseline freezes known contract incompleteness in the retained legacy diff --git a/doc/plan/active__legacy-task-migration-map.md b/doc/plan/active__legacy-task-migration-map.md index 845bd8ff..69a26925 100644 --- a/doc/plan/active__legacy-task-migration-map.md +++ b/doc/plan/active__legacy-task-migration-map.md @@ -155,7 +155,7 @@ remain outside executable pricing selection. | `T125` | `proof_only_hold` | `TASKS_PROOF_LEGACY.yaml` | Swing option: dynamic programming on tree vs LSM MC | | `T126` | `proof_only_hold` | `TASKS_PROOF_LEGACY.yaml` | Spread option (Kirk approximation) vs 2D MC vs 2D FFT | | `E21` | `proof_only_hold` | `TASKS_PROOF_LEGACY.yaml` | European equity call: 5-way (tree, PDE, MC, FFT, COS) | -| `E22` | `proof_only_hold` | `TASKS_PROOF_LEGACY.yaml` | Cap/floor: Black caplet stack vs MC rate simulation | +| `E22` | `executable_pricing` | `TASKS_PROOF_LEGACY.yaml` | Authored Black caplets versus seeded lognormal forward marginals | | `E23` | `proof_only_hold` | `TASKS_PROOF_LEGACY.yaml` | European equity call under local vol: PDE vs MC | | `E24` | `proof_only_hold` | `TASKS_PROOF_LEGACY.yaml` | Merton jump-diffusion call: MC vs FFT | | `E25` | `proof_only_hold` | `TASKS_PROOF_LEGACY.yaml` | FX option (EURUSD): GK analytical vs MC | diff --git a/doc/plan/active__task-manifest-integrity.md b/doc/plan/active__task-manifest-integrity.md index ade0bae2..ae5ffcb2 100644 --- a/doc/plan/active__task-manifest-integrity.md +++ b/doc/plan/active__task-manifest-integrity.md @@ -37,7 +37,7 @@ exact capability blocker before Trellis can synthesize economic inputs. | 10.1 | `QUA-1255` | Done | T02/T17 receive authored callable pricing fixtures | | 10.2 | `QUA-1258` | Blocked | T82/T89 receive explicit callable analytics outputs | | 10.3 | `QUA-1254` | Backlog | T73 receives an authored European swaption contract | -| 10.4 | `QUA-1256` | Backlog | E22 receives an authored cap-strip contract | +| 10.4 | `QUA-1256` | In Progress | E22 receives an authored cap-strip contract | | 10.5 | `QUA-1257` | Done | T102 receives an authored terminal-basket contract | | 10.6 | `QUA-1259` | Blocked | Remove the repaired title-derived pricing bootstraps | | 11 | `QUA-1251` | Blocked | Reusable variable-coupon callable-bond primitive | diff --git a/docs/developer/implementation_journey_prompt_to_price.md b/docs/developer/implementation_journey_prompt_to_price.md index 8bdf8ffa..bb3f0147 100644 --- a/docs/developer/implementation_journey_prompt_to_price.md +++ b/docs/developer/implementation_journey_prompt_to_price.md @@ -182,6 +182,22 @@ It was: ## What Is Still Transitional +The E22 cap-strip proof now follows an authored contract end to end. Its +manifest specifies every accrual, fixing and payment date, ACT/360 accrual, +ACT/365 option time, ACT/ACT ISDA curve time, unadjusted dates, named market scenario and the +100,000-path/seed-42 antithetic forward-marginal comparison. The direct +semantic bridge preserves these fields when specializing to analytical or +Monte Carlo. Generated spec defaults and benchmark overrides carry the +same dates and controls; optional strip terms keep their declared defaults +so smoke validation cannot invent a callable feature or collar strike. + +The exact validator accepts the authored source row and its canonical +materialized market envelope. It rejects altered economics, controls or +scenario values before build. The title is descriptive only, and the +comparison reports USD holder present value against its explicit 0.5% +analytical-reference tolerance. Reproduce it with +`scripts/run_tasks.py --task-id E22 --corpus proof_legacy --status all --offline-local-agents --fresh-build`. + The system is much more coherent than it was, but a few things remain transitional: diff --git a/docs/quant/pricing_stack.rst b/docs/quant/pricing_stack.rst index b9321439..30b03fa7 100644 --- a/docs/quant/pricing_stack.rst +++ b/docs/quant/pricing_stack.rst @@ -1282,6 +1282,27 @@ caplet/floorlet strips no longer get rejected as generic ``automatic`` event routes when the lowered family IR is already the checked analytical strip surface. +The authored E22 cap-strip proof uses the existing +``price_rate_cap_floor_strip_analytical`` and +``price_rate_cap_floor_strip_monte_carlo`` kernels. Both sum the same twenty +quarterly caplets on USD 1,000,000 at a 4% strike, from 2025-02-15 through +2030-02-15. The contract authors every accrual boundary, fixing date and +payment date. Fixings occur at accrual start, payments at accrual end, +with zero lag and unadjusted dates; ACT/360 measures accrual and ACT/365 +measures option expiry. Both date-aware curves use the separately authored +ACT/ACT ISDA clock for discount factors and projected forwards. The named ``usd_rates_smile`` +scenario supplies 4% OIS discounting, a 4.25% SOFR-3M forecast curve and +20% Black forward volatility, valued on 2024-11-15. + +The Monte Carlo leg samples each caplet's lognormal forward marginal with +100,000 antithetic samples and seed 42. It uses the same forward, fixing +expiry, accrual, payment discount and Black volatility as the analytical +leg. The 0.5% tolerance is a relative error in holder present value. This +proof does not simulate a short-rate path or assert a joint model for +caplet forwards; cross-caplet dependence is unnecessary for this linear +sum of individual option expectations. It does not validate calibration, +shifted/normal volatility, seasoned fixings or production holiday rules. + Below those public callable wrappers, the reusable coupon/event/control layer now lives in ``trellis.models.short_rate_fixed_income``. Coupon schedule compilation, embedded issuer/holder exercise semantics, straight-bond diff --git a/docs/user_guide/pricing.rst b/docs/user_guide/pricing.rst index 0718617c..fcd710c8 100644 --- a/docs/user_guide/pricing.rst +++ b/docs/user_guide/pricing.rst @@ -771,6 +771,16 @@ lanes consume the same declared contract, and the task carries its own 2% comparison tolerance. Trellis does not infer those terms from the task title; missing, extra, or changed fields are rejected before pricing. +E22 is an authored USD cap-strip example. It fixes the notional, strike, +all quarterly accrual/fixing/payment dates, date conventions, named forecast +and discount curves, Black volatility, sample count and random seed. +Its analytical reference is the discounted Black caplet sum; its Monte +Carlo comparison samples the same lognormal caplet forwards and allows +0.5% relative price error. This is a forward-marginal pricing proof, so it +does not establish short-rate path simulation or general cap-market +calibration support. The manifest fails admission if its required terms +or controls are missing or changed. + They also now carry compiler-emitted lane obligations. In practice that means the build loop sees the computational lane first (analytical, lattice, Monte Carlo, PDE, and so on), the timeline and market bindings that lane requires, diff --git a/tests/test_tasks/test_e22_cap_strip.py b/tests/test_tasks/test_e22_cap_strip.py new file mode 100644 index 00000000..bd02356c --- /dev/null +++ b/tests/test_tasks/test_e22_cap_strip.py @@ -0,0 +1,410 @@ +from __future__ import annotations + +from copy import deepcopy +from datetime import date +from types import SimpleNamespace + +import pytest + + +def _task(): + from trellis.agent.task_manifests import load_task_manifest + + return next( + t for t in load_task_manifest("TASKS_PROOF_LEGACY.yaml") if t["id"] == "E22" + ) + + +def test_e22_authored_contract_is_admitted_and_title_independent(): + from trellis.agent.task_manifest_validation import assert_executable_task_selection + from trellis.agent.task_runtime import ( + benchmark_spec_overrides, + task_to_semantic_contract, + ) + + task = _task() + assert_executable_task_selection([task]) + renamed = {**task, "title": "Uninformative title"} + assert_executable_task_selection([renamed]) + assert benchmark_spec_overrides(task) == benchmark_spec_overrides(renamed) + original = task_to_semantic_contract(task) + changed = task_to_semantic_contract(renamed) + assert original.product == changed.product + assert original.product.term_fields["n_paths"] == 100_000 + assert original.product.term_fields["model_time_day_count"] == "ACT/365" + assert original.product.timeline.settlement_dates == tuple( + task["benchmark_contract"]["payment_dates"] + ) + + +def test_e22_structured_bridge_and_method_rebuild_preserve_the_contract(monkeypatch): + from trellis.agent.semantic_contracts import specialize_semantic_contract_for_method + from trellis.agent.task_runtime import ( + _effective_task_description, + task_to_semantic_contract, + ) + + def forbidden(*args, **kwargs): + raise AssertionError("Authored E22 must not be synthesized from prose") + + monkeypatch.setattr( + "trellis.agent.semantic_contracts.draft_semantic_contract", forbidden + ) + monkeypatch.setattr( + "trellis.agent.task_runtime._bootstrap_rate_cap_floor_description", forbidden + ) + task = _task() + assert _effective_task_description(task) == _effective_task_description( + {**task, "title": "Changed"} + ) + contract = task_to_semantic_contract(task) + rebuilt = specialize_semantic_contract_for_method( + contract, preferred_method="monte_carlo" + ) + assert dict(rebuilt.product.term_fields) == dict(contract.product.term_fields) + assert rebuilt.product.timeline == contract.product.timeline + + +def test_e22_generated_schema_keeps_explicit_schedule_and_controls(): + from trellis.agent.executor import ( + _generate_skeleton, + _hydrate_spec_schema_defaults_from_semantics, + ) + from trellis.agent.planner import STATIC_SPECS + from trellis.agent.task_runtime import task_to_semantic_contract + + schema = _hydrate_spec_schema_defaults_from_semantics( + STATIC_SPECS["period_rate_option_strip"], + semantic_contract=task_to_semantic_contract(_task()), + ) + defaults = {field.name: field.default for field in schema.fields} + assert defaults["n_paths"] == "100000" + assert defaults["seed"] == "42" + assert defaults["day_count"] == "DayCountConvention.ACT_360" + assert defaults["frequency"] == "Frequency.QUARTERLY" + assert "date(2025, 2, 15)" in defaults["fixing_dates"] + assert "date(2030, 2, 15)" in defaults["payment_dates"] + generated = _generate_skeleton(schema, "E22 authored cap") + compile(generated, "", "exec") + + +def test_cap_strip_smoke_fixture_does_not_invent_callable_or_collar_terms(): + from trellis.agent.executor import _generate_skeleton, _make_test_payoff + from trellis.agent.planner import STATIC_SPECS + + schema = STATIC_SPECS["period_rate_option_strip"] + namespace = {} + exec(_generate_skeleton(schema, "Non-callable cap"), namespace) + payoff = _make_test_payoff(namespace[schema.class_name], schema, date(2024, 11, 15)) + assert payoff._spec.call_price is None + assert payoff._spec.exercise_dates is None + assert payoff._spec.cap_strike is None + assert payoff._spec.floor_strike is None + + +def test_e22_admits_materialized_market_and_rejects_market_override(): + from trellis.agent.market_scenarios import market_scenario_contract_from_task + from trellis.agent.task_manifest_validation import ( + TaskManifestValidationError, + assert_executable_task_selection, + ) + + task = _task() + scenario = market_scenario_contract_from_task(task) + task["market"] = { + "source": scenario.source, + "as_of": scenario.as_of.isoformat(), + **dict(scenario.selected_components), + "scenario_contract": scenario.to_payload(), + "scenario_digest": scenario.scenario_digest, + "scenario_schema_version": scenario.schema_version, + "scenario_constructor_kind": scenario.constructor_kind, + "benchmark_inputs": scenario.financepy_inputs(), + } + assert_executable_task_selection([task]) + task["market"]["benchmark_inputs"]["black_vol"] = 0.9 + with pytest.raises( + TaskManifestValidationError, match="legacy.cap_strip_invalid_contract" + ): + assert_executable_task_selection([task]) + + +@pytest.mark.parametrize( + "key", ["notional", "payment_dates", "fixing_rule", "model_time_day_count", "seed"] +) +def test_e22_rejects_missing_authored_terms(key): + from trellis.agent.task_manifest_validation import ( + TaskManifestValidationError, + assert_executable_task_selection, + ) + + task = _task() + del task["benchmark_contract"][key] + with pytest.raises( + TaskManifestValidationError, match="legacy.cap_strip_invalid_contract" + ): + assert_executable_task_selection([task]) + + +def test_e22_validates_referenced_scenario_at_the_requested_root(tmp_path): + from pathlib import Path + + import yaml + + from trellis.agent.task_manifest_validation import ( + TaskManifestValidationError, + assert_executable_task_selection, + ) + + source = Path(__file__).resolve().parents[2] / "MARKET_SCENARIOS.yaml" + payload = yaml.safe_load(source.read_text()) + payload["scenarios"]["usd_rates_smile"]["constructor"]["black_vol"] = 0.3 + (tmp_path / "MARKET_SCENARIOS.yaml").write_text(yaml.safe_dump(payload)) + with pytest.raises( + TaskManifestValidationError, match="legacy.cap_strip_invalid_contract" + ): + assert_executable_task_selection([_task()], root=tmp_path) + + +@pytest.mark.parametrize("field", ["seed", "simulation_seed"]) +def test_e22_rejects_top_level_seed_aliases_that_conflict_with_authored_controls(field): + from trellis.agent.task_manifest_validation import ( + TaskManifestValidationError, + assert_executable_task_selection, + ) + + task = {**_task(), field: 7} + with pytest.raises( + TaskManifestValidationError, match="legacy.cap_strip_invalid_contract" + ): + assert_executable_task_selection([task]) + + +@pytest.mark.parametrize( + "section,key,value", + [ + ("benchmark_contract", "strike", 0.05), + ("benchmark_contract", "n_paths", 20_000), + ("benchmark_contract", "seed", 7), + ("benchmark_contract", "n_steps", 64), + ("benchmark_contract", "model_time_day_count", "ACT/360"), + ("benchmark_contract", "business_day_adjustment", "following"), + ("benchmark_contract", "fixing_dates", ["2025-02-17"]), + ("cross_validate", "tolerance_pct", 5.0), + ], +) +def test_e22_rejects_drift_and_unexercised_controls(section, key, value): + from trellis.agent.task_manifest_validation import ( + TaskManifestValidationError, + assert_executable_task_selection, + ) + + task = deepcopy(_task()) + task[section][key] = value + with pytest.raises(TaskManifestValidationError) as exc: + assert_executable_task_selection([task]) + assert "legacy.cap_strip_invalid_contract" in str(exc.value) + + +def test_e22_generated_monte_carlo_exercises_authored_controls(): + from trellis.agent.executor import _deterministic_exact_binding_evaluate_body + + plan = SimpleNamespace( + method="monte_carlo", + instrument_type="period_rate_option_strip", + lane_exact_binding_refs=( + "trellis.models.rate_cap_floor.price_rate_cap_floor_strip_monte_carlo", + ), + backend_helper_refs=(), + primitive_plan=None, + ) + body = _deterministic_exact_binding_evaluate_body( + plan, comparison_target="monte_carlo" + ) + assert body + captured = {} + + def record(**kwargs): + captured.update(kwargs) + return 123.0 + + namespace = {"price_rate_cap_floor_strip_monte_carlo": record} + exec( + "def evaluate(self, market_state):\n" + + "\n".join(" " + line for line in body.splitlines()), + namespace, + ) + spec = SimpleNamespace(n_paths=1234, seed=17) + assert namespace["evaluate"](SimpleNamespace(_spec=spec), object()) == 123.0 + assert captured["n_paths"] == 1234 + assert captured["seed"] == 17 + + +def test_e22_authored_black_caplets_match_seeded_forward_marginals(): + from trellis.agent.task_runtime import ( + benchmark_spec_overrides, + build_market_state_for_task, + ) + from trellis.models.rate_cap_floor import ( + price_rate_cap_floor_strip_analytical, + price_rate_cap_floor_strip_monte_carlo, + ) + from trellis.core.types import DayCountConvention + + task = _task() + market, _ = build_market_state_for_task(task) + assert market.discount.curve_day_count == DayCountConvention.ACT_ACT_ISDA + assert ( + market.forecast_curves["USD-SOFR-3M"].curve_day_count + == DayCountConvention.ACT_ACT_ISDA + ) + overrides = benchmark_spec_overrides(task) + assert len(overrides["accrual_dates"]) == 21 + assert overrides["fixing_dates"] == overrides["accrual_dates"][:-1] + assert overrides["payment_dates"] == overrides["accrual_dates"][1:] + assert overrides["start_date"] == date(2025, 2, 15) + keys = ( + "notional", + "strike", + "start_date", + "end_date", + "frequency", + "day_count", + "rate_index", + "accrual_dates", + "fixing_dates", + "calendar_name", + "business_day_adjustment", + ) + kwargs = {k: overrides[k] for k in keys} + kwargs["coupon_dates"] = overrides["payment_dates"] + analytical = price_rate_cap_floor_strip_analytical(market, **kwargs) + sampled = price_rate_cap_floor_strip_monte_carlo( + market, **kwargs, n_paths=overrides["n_paths"], seed=overrides["seed"] + ) + assert analytical == pytest.approx(27531.179158915023, rel=1e-12) + assert sampled == pytest.approx(27538.523575656935, rel=1e-12) + assert ( + abs(sampled / analytical - 1.0) * 100 < task["cross_validate"]["tolerance_pct"] + ) + assert sampled == price_rate_cap_floor_strip_monte_carlo( + market, **kwargs, n_paths=overrides["n_paths"], seed=overrides["seed"] + ) + + +@pytest.mark.global_workflow +def test_e22_loaded_contract_prices_both_declared_lanes_end_to_end( + monkeypatch, tmp_path +): + """Defend the authored contract across compilation, hydration and comparison.""" + import sys + from pathlib import Path + + import trellis.agent.analytical_traces as analytical_traces + import trellis.agent.executor as executor + import trellis.agent.model_audit as model_audit + import trellis.agent.platform_requests as platform_requests + import trellis.agent.platform_traces as platform_traces + from trellis.agent.offline_agents import offline_local_agent_run_scope + from trellis.agent.task_runtime import run_task + from trellis.engine.payoff_pricer import price_payoff + + task = {**_task(), "title": "Uninformative label"} + observed = [] + + def observed_price(payoff, market): + observed.append((payoff._spec, market.settlement)) + return price_payoff(payoff, market) + + def write_generated_module(module_path: str, code: str) -> Path: + output_path = tmp_path / "generated" / module_path + output_path.parent.mkdir(parents=True, exist_ok=True) + output_path.write_text(code) + return output_path + + monkeypatch.setattr(executor, "REPO_ROOT", tmp_path) + monkeypatch.setattr(executor, "TRELLIS_PACKAGE_ROOT", tmp_path / "trellis") + monkeypatch.setattr(executor, "_REPO_REVISION", "test") + monkeypatch.setattr(executor, "write_module", write_generated_module) + monkeypatch.setattr( + analytical_traces, "TRACE_ROOT", tmp_path / "traces" / "analytical" + ) + monkeypatch.setattr(platform_traces, "TRACE_ROOT", tmp_path / "traces" / "platform") + monkeypatch.setattr(model_audit, "_AUDIT_DIR", tmp_path / "audits") + monkeypatch.setattr( + platform_requests, "_record_semantic_extension_artifact", lambda *a, **kw: None + ) + monkeypatch.setenv("TRELLIS_SKIP_TASK_DIAGNOSIS_PERSIST", "1") + monkeypatch.setenv("TRELLIS_SKIP_POST_BUILD_REFLECTION", "1") + monkeypatch.setenv("TRELLIS_SKIP_POST_BUILD_CONSOLIDATION", "1") + + module_name = "trellis.instruments._agent._fresh.periodrateoptionstrip" + previous_module = sys.modules.get(module_name) + try: + with offline_local_agent_run_scope(): + result = run_task( + task, + market_state=None, + price_fn=observed_price, + fresh_build=True, + recovery_mode="strict", + execution_mode_override="deterministic_replay", + task_run_storage_root=tmp_path / "task-runs", + task_run_storage_layout="standalone", + ) + finally: + if previous_module is None: + sys.modules.pop(module_name, None) + else: + sys.modules[module_name] = previous_module + + assert result["success"] is True, result.get("failures") + assert result["passed_expectation"] is True + assert result["attempts"] == 0 + assert result["token_usage_summary"]["call_count"] == 0 + assert result["instrument_type"] == "cap" + assert result["runtime_contract"]["simulation_seed"] == 42 + assert ( + result["runtime_contract"]["simulation_identity"]["seed_source"] + == "task.benchmark_contract.seed" + ) + comparison = result["cross_validation"] + assert comparison["status"] == "passed" + assert comparison["prices"] == pytest.approx( + { + "analytical": 27531.179158915023, + "monte_carlo": 27538.523575656935, + }, + rel=1e-12, + ) + assert comparison["reference_target"] == "analytical" + assert comparison["tolerance_pct"] == 0.5 + assert comparison["failed_targets"] == [] + for target in ("analytical", "monte_carlo"): + binding = comparison["artifact_coherence"][target] + assert binding["status"] == "bound_unique_artifact" + assert ( + binding["selected_semantic_axes"]["payoff_family"] + == "period_rate_option_strip" + ) + acceptance = comparison["target_acceptance"][target] + assert acceptance["output_currency"] == "USD" + assert acceptance["output_unit"] == "currency_amount" + assert acceptance["tolerance_unit"] == "percent_of_reference_price" + assert acceptance["tolerance_pct"] == 0.5 + assert comparison["artifact_coherence"]["monte_carlo"][ + "exercised_spec_overrides" + ] == { + "n_paths": 100000, + "seed": 42, + } + assert observed + for spec, settlement in observed: + assert settlement == date(2024, 11, 15) + assert spec.notional == 1000000.0 + assert spec.strike == 0.04 + assert spec.n_paths == 100000 + assert spec.seed == 42 + assert len(spec.accrual_dates) == 21 + assert spec.fixing_dates == spec.accrual_dates[:-1] + assert spec.payment_dates == spec.accrual_dates[1:] diff --git a/trellis/agent/benchmark_contracts.py b/trellis/agent/benchmark_contracts.py index b31aba9e..a50da3f5 100644 --- a/trellis/agent/benchmark_contracts.py +++ b/trellis/agent/benchmark_contracts.py @@ -277,6 +277,8 @@ def benchmark_request_description( title = "USD fixed-coupon callable bond proof" elif str(task.get("id") or "").strip() == "T102" and product == "rainbow_option": title = "Two-asset European terminal best-of call" + elif str(task.get("id") or "").strip() == "E22" and product == "period_rate_option_strip": + title = "USD cap strip with authored forward-marginal comparison" else: title = str(task.get("title") or "Benchmark pricing task").strip() lines = [f"Build a pricer for: {title}", ""] @@ -922,6 +924,18 @@ def _benchmark_detail_lines( lines.append(f"Payment frequency: {contract['payment_frequency']}.") if contract.get("day_count"): lines.append(f"Day count: {contract['day_count']}.") + for field in ( + "model_time_day_count", "discount_curve_day_count", "forecast_curve_day_count", + "calendar_name", "business_day_adjustment", + "fixing_rule", "payment_rule", "fixing_lag_days", "payment_lag_days", + "mc_distribution", "sampling", "n_paths", "seed", + ): + if contract.get(field) is not None: + lines.append(f"{field}: {contract[field]}.") + for field in ("accrual_dates", "fixing_dates", "payment_dates"): + if contract.get(field): + dates = ", ".join(str(value) for value in contract[field]) + lines.append(f"{field}: {dates}.") if scenario_contract is not None and scenario_contract.forecast_curve_name: lines.append(f"Rate index: {scenario_contract.forecast_curve_name}.") model = str(contract.get("model") or "").strip().lower() @@ -1252,6 +1266,8 @@ def _rate_cap_floor_overrides( or ("collar" if product == "rate_cap_floor_collar" else None) ), "model": str(contract.get("model") or "black").strip().lower() or None, + "n_paths": int(contract["n_paths"]) if contract.get("n_paths") is not None else None, + "seed": int(contract["seed"]) if contract.get("seed") is not None else None, "shift": _float_or_none(contract.get("shift")), "sabr": dict(contract.get("sabr") or {}) or None, "exercise_style": str(contract.get("style") or "").strip().lower() or None, diff --git a/trellis/agent/executor.py b/trellis/agent/executor.py index 336cf71e..bb3fd800 100644 --- a/trellis/agent/executor.py +++ b/trellis/agent/executor.py @@ -628,6 +628,34 @@ def _enum_default(prefix: str, raw_value: object | None) -> str | None: rate_index = term_fields.get("rate_index") if rate_index not in {None, ""}: overrides["rate_index"] = str(rate_index).strip() + elif spec_name == "PeriodRateOptionStripSpec": + def _date_default(value: object) -> str: + parsed = value if isinstance(value, date) else date.fromisoformat(str(value)) + return f"date({parsed.year}, {parsed.month}, {parsed.day})" + + for name in ("notional", "strike", "n_paths", "seed"): + if term_fields.get(name) is not None: + overrides[name] = repr(term_fields[name]) + for name in ("start_date", "end_date"): + if term_fields.get(name) is not None: + overrides[name] = _date_default(term_fields[name]) + for name in ("accrual_dates", "fixing_dates", "payment_dates"): + if term_fields.get(name): + dates = ", ".join(_date_default(value) for value in term_fields[name]) + overrides[name] = f"({dates},)" + for name in ("rate_index", "calendar_name", "business_day_adjustment", "model"): + if term_fields.get(name) is not None: + overrides[name] = str(term_fields[name]) + if term_fields.get("cap_floor") is not None: + overrides["instrument_class"] = str(term_fields["cap_floor"]) + if term_fields.get("payment_frequency") is not None: + overrides["frequency"] = _enum_default( + "Frequency", str(term_fields["payment_frequency"]).upper(), + ) + if term_fields.get("day_count") is not None: + overrides["day_count"] = _enum_default( + "DayCountConvention", str(term_fields["day_count"]).replace("/", "_").upper(), + ) elif spec_name == "NthToDefaultSpec": reference_names = tuple(getattr(product, "constituents", ()) or ()) basket_weights = tuple(term_fields.get("basket_weights", ()) or ()) @@ -3810,6 +3838,13 @@ def _make_test_payoff( name_defaults["spot"] = spot_default name_defaults["strike"] = spot_default + if spec_schema.spec_name == "PeriodRateOptionStripSpec": + # Optional strip terms are economic choices: synthetic call/collar + # values would turn a plain cap into a different, unsupported product. + for field in spec_schema.fields: + if field.default is not None: + name_defaults.pop(field.name, None) + description = getattr(payoff_cls, "__doc__", "") or getattr(module, "__doc__", "") or "" description_defaults = _description_spec_defaults( spec_schema, @@ -7808,8 +7843,8 @@ def evaluate_state(state): ), exercise_dates=getattr(spec, "exercise_dates", None) or getattr(spec, "call_dates", None), is_payer=getattr(spec, "is_payer", None), - n_paths=20000, - seed=42, + n_paths=getattr(spec, "n_paths", 20000), + seed=getattr(spec, "seed", 42), notional=getattr(spec, "notional", None), strike=( getattr(spec, "strike", None) @@ -7870,8 +7905,8 @@ def evaluate_state(state): frequency=spec.frequency, day_count=spec.day_count, rate_index=spec.rate_index, - n_paths=20000, - seed=42, + n_paths=getattr(spec, "n_paths", 20000), + seed=getattr(spec, "seed", 42), ) """ ).rstrip() diff --git a/trellis/agent/planner.py b/trellis/agent/planner.py index 9e8cdb70..f4d591d4 100644 --- a/trellis/agent/planner.py +++ b/trellis/agent/planner.py @@ -155,6 +155,8 @@ class BuildPlan: FieldDef("rate_index", "str | None", "Forecast curve key", "None"), FieldDef("calendar_name", "str | None", "Calendar name for generated schedules", "None"), FieldDef("business_day_adjustment", "str | None", "Business-day adjustment convention", "None"), + FieldDef("n_paths", "int", "Monte Carlo samples per caplet", "20000"), + FieldDef("seed", "int", "Monte Carlo random seed", "42"), FieldDef("model", "str | None", "Cap/floor model override", "None"), FieldDef("shift", "float | None", "Shift for shifted-Black pricing", "None"), FieldDef("sabr", "dict[str, float] | None", "SABR parameter bundle", "None"), diff --git a/trellis/agent/semantic_contracts.py b/trellis/agent/semantic_contracts.py index 6725d649..0b9e9556 100644 --- a/trellis/agent/semantic_contracts.py +++ b/trellis/agent/semantic_contracts.py @@ -4190,6 +4190,7 @@ def make_period_rate_option_strip_contract( instrument_class: str, observation_schedule: tuple[str, ...] | list[str], preferred_method: str = "analytical", + term_fields: Mapping[str, Any] | None = None, ) -> SemanticContract: """Construct a schedule-driven period rate-option strip semantic contract.""" normalized_instrument = str(instrument_class or "").strip().lower() @@ -4224,7 +4225,7 @@ def make_period_rate_option_strip_contract( option_type=option_type, timeline=_default_semantic_timeline( schedule, - settlement_dates=schedule, + settlement_dates=tuple((term_fields or {}).get("payment_dates") or schedule), state_update_dates=schedule, ), underlier_structure="single_curve_rate_style", @@ -4290,6 +4291,7 @@ def make_period_rate_option_strip_contract( ), term_fields=_freeze_mapping( { + **dict(term_fields or {}), "option_type": option_type, "schedule_authority": schedule_authority, } @@ -5489,6 +5491,7 @@ def _rebuild_period_rate_option_strip_contract( instrument_class=str(getattr(product, "instrument_class", "") or "cap"), observation_schedule=tuple(getattr(product, "observation_schedule", ()) or ()), preferred_method=normalized_method, + term_fields=dict(product.term_fields), ) diff --git a/trellis/agent/task_manifest_validation.py b/trellis/agent/task_manifest_validation.py index 66c5c55f..84e9ac4f 100644 --- a/trellis/agent/task_manifest_validation.py +++ b/trellis/agent/task_manifest_validation.py @@ -1562,6 +1562,13 @@ def _validate_legacy_task( ) ) + if task_id == "E22": + issues.extend( + _validate_legacy_cap_strip_comparison_contract( + manifest_name, task, path, root=root, + ) + ) + if task_id in {"T02", "T17"}: issues.extend( _validate_legacy_callable_bond_comparison_contract( @@ -1645,6 +1652,134 @@ def _validate_legacy_task( return issues +def _validate_legacy_cap_strip_comparison_contract( + manifest_name: str, + task: Mapping[str, Any], + path: str, + *, + root: Path | None = None, +) -> list[TaskManifestIssue]: + """Admit only the authored unadjusted Black-forward E22 proof.""" + from trellis.agent.market_scenarios import load_market_scenario_contracts + + boundaries = [ + date(2025 + (1 + 3 * i) // 12, (1 + 3 * i) % 12 + 1, 15).isoformat() + for i in range(21) + ] + expected_contract = { + "product": "period_rate_option_strip", "cap_floor": "cap", + "currency": "USD", "notional": 1000000.0, "strike": 0.04, + "valuation_date": "2024-11-15", + "start_date": "2025-02-15", "end_date": "2030-02-15", + "payment_frequency": "quarterly", "day_count": "ACT/360", + "model_time_day_count": "ACT/365", "calendar_name": "weekend_only", + "discount_curve_day_count": "ACT/ACT ISDA", + "forecast_curve_day_count": "ACT/ACT ISDA", + "business_day_adjustment": "unadjusted", + "fixing_rule": "accrual_start", "payment_rule": "accrual_end", + "fixing_lag_days": 0, "payment_lag_days": 0, + "accrual_dates": boundaries, "fixing_dates": boundaries[:-1], + "payment_dates": boundaries[1:], "rate_index": "USD-SOFR-3M", + "model": "black", + "mc_distribution": "independent_lognormal_forward_marginals", + "sampling": "antithetic", "n_paths": 100000, "seed": 42, + "valuation_measure": "holder_present_value", + "output_unit": "currency_amount", "output_currency": "USD", + } + axes = { + "payoff_family": "period_rate_option_strip", "exercise_style": "none", + "model_family": "interest_rate", "observation_style": "fixed_schedule", + } + expected_comparison = { + "internal": ["analytical", "monte_carlo"], "reference_target": "analytical", + "relations": {"monte_carlo": "within_tolerance"}, "tolerance_pct": 0.5, + "tolerance_unit": "percent_of_reference_price", + "output_unit": "currency_amount", "output_currency": "USD", + "target_contracts": { + "analytical": { + "method": "analytical", **axes, "variant_parameters": {"model": "black"}, + "route_family": "analytical", + "backend_binding_id": "trellis.models.rate_cap_floor.price_rate_cap_floor_strip_analytical", + "validation_bundle_id": "analytical:cap", + }, + "monte_carlo": { + "method": "monte_carlo", **axes, + "route_family": "monte_carlo", + "backend_binding_id": "trellis.models.rate_cap_floor.price_rate_cap_floor_strip_monte_carlo", + "validation_bundle_id": "monte_carlo:cap", + "variant_parameters": { + "distribution": "independent_lognormal_forward_marginals", + "sampling": "antithetic", + }, + "spec_overrides": {"n_paths": 100000, "seed": 42}, + }, + }, + } + selected = { + "discount_curve": "usd_ois", "forecast_curve": "USD-SOFR-3M", + "vol_surface": "usd_rates_smile", + } + scenario_valid = canonical_market_matches = False + try: + scenarios = load_market_scenario_contracts(**({"root": root} if root else {})) + scenario = scenarios.get("usd_rates_smile") + scenario_valid = bool( + scenario is not None and scenario.source == "mock" + and scenario.as_of.isoformat() == "2024-11-15" + and scenario.valuation_date == date(2024, 11, 15) + and scenario.constructor_kind == "flat_rates" + and scenario.domestic_rate == 0.04 and scenario.forecast_rate == 0.0425 + and scenario.forecast_curve_name == "USD-SOFR-3M" + and scenario.black_vol == 0.2 + and dict(scenario.selected_components) == selected + ) + market = task.get("market") + canonical_market_matches = market is None + if scenario is not None and isinstance(market, Mapping): + expected_market = { + "source": scenario.source, "as_of": scenario.as_of.isoformat(), + **dict(scenario.selected_components), + "scenario_contract": scenario.to_payload(), + "scenario_digest": scenario.scenario_digest, + "scenario_schema_version": scenario.schema_version, + "scenario_constructor_kind": scenario.constructor_kind, + "benchmark_inputs": scenario.financepy_inputs(), + } + canonical_market_matches = _manifest_value_matches(market, expected_market) + except (OSError, ValueError, yaml.YAMLError): + pass + expected_description = ( + "Price the authored USD cap strip under the named rates scenario. Compare " + "discounted Black-76 caplets with seeded antithetic Monte Carlo of independent " + "lognormal caplet forward marginals, reporting holder present value in USD." + ) + valid = all(( + task.get("task_disposition") == "executable_pricing", + task.get("instrument_type") == "period_rate_option_strip", + task.get("market_scenario_id") == "usd_rates_smile", + task.get("validation_policy") == "invariants_and_cross_method", + task.get("description") == expected_description, + _manifest_value_matches(task.get("construct"), ["analytical", "monte_carlo"]), + _manifest_value_matches(task.get("benchmark_contract"), expected_contract), + _manifest_value_matches(task.get("cross_validate"), expected_comparison), + _manifest_value_matches(task.get("market_assertions"), { + "requires": ["discount_curve", "forward_curve", "black_vol_surface"], + "selected": selected, + }), + scenario_valid, canonical_market_matches, + not any(k in task for k in ( + "comparison_regime", "financepy_binding_id", "proof_fixture_id", + "expected_outcome", "expected_blocker_ids", "honest_block_contract", + "seed", "simulation_seed", + )), + )) + return [] if valid else [_issue( + manifest_name, "legacy.cap_strip_invalid_contract", + "E22 requires the exact authored Black caplet and sampled-forward proof contract", + task_id=_text(task.get("id")), path=path, + )] + + def _validate_legacy_callable_bond_comparison_contract( manifest_name: str, task: Mapping[str, Any], diff --git a/trellis/agent/task_runtime.py b/trellis/agent/task_runtime.py index 2a447032..b56addb2 100644 --- a/trellis/agent/task_runtime.py +++ b/trellis/agent/task_runtime.py @@ -1584,6 +1584,20 @@ def _proof_legacy_semantic_contract(task: dict, description: str): }, ) + if task_id == "E22": + from trellis.agent.semantic_contracts import make_period_rate_option_strip_contract + + contract = task.get("benchmark_contract") + if not isinstance(contract, Mapping): + raise ValueError("E22 requires a structured benchmark_contract") + return make_period_rate_option_strip_contract( + description=description, + instrument_class=str(contract["cap_floor"]), + observation_schedule=tuple(str(value) for value in contract["fixing_dates"]), + preferred_method="analytical", + term_fields=dict(contract), + ) + if task_id == "T102": from trellis.agent.semantic_contracts import make_terminal_basket_option_contract From ed0f2adb09696f6e025347ad95cd6159538652ca Mon Sep 17 00:00:00 2001 From: steveya Date: Sun, 13 Sep 2026 00:15:51 -0400 Subject: [PATCH 2/3] Align cap-strip proof metadata and repair review regressions --- LIMITATIONS.md | 2 +- .../implementation_journey_prompt_to_price.md | 12 +++- docs/quant/pricing_stack.rst | 5 +- docs/user_guide/pricing.rst | 4 +- tests/evals/binding_first_exotic_proof.yaml | 6 +- tests/evals/stress_tasks.yaml | 6 +- tests/test_agent/test_family_lowering_ir.py | 21 ++++--- tests/test_tasks/test_e22_cap_strip.py | 58 ++++++++++++++++++- trellis/agent/benchmark_contracts.py | 1 + trellis/agent/executor.py | 5 ++ trellis/agent/family_lowering_ir.py | 45 ++++++++++++-- 11 files changed, 142 insertions(+), 23 deletions(-) diff --git a/LIMITATIONS.md b/LIMITATIONS.md index a61b3e7f..163e6d53 100644 --- a/LIMITATIONS.md +++ b/LIMITATIONS.md @@ -70,7 +70,7 @@ ground truth until revalidated. | L49 | **Agent-cycle and institutional approval evidence are governed internal review surfaces, not external model certification** — Trellis now exposes a stable `agent_cycle` result surface for quant/critic/arbiter/model-validator evidence, model-promotion eligibility, and benchmark trigger rates, and `policy_bundle.production.institutional` can require approval, model-review, snapshot, run-artifact, and audit-bundle evidence before production execution, but these surfaces only certify recorded internal governance evidence for the run or model version | Product and desk review surfaces can show why a cycle passed, failed, was unavailable, or lacked required institutional approval artifacts, but they must not be read as external model approval, regulatory sign-off, xVA/FpML coverage, or correctness beyond the recorded validation scope | `trellis/agent/cycle_surface.py`, `trellis/agent/task_runtime.py`, `trellis/platform/models.py`, `trellis/platform/policies.py`, `trellis/platform/services/pricing_service.py`, `docs/developer/audit_and_observability.rst`, `docs/developer/hosting_and_configuration.rst`, `docs/user_guide/pricing.rst` | | L57 | **Typed comparison-target coherence exposes rather than fills missing numerical composition** — Explicit targets carry canonical method, route/binding, variant, validation, and semantic identities, and the runtime rejects partial explicit target sets, ambiguous references, missing declarations, and unbound shared artifacts before pricing. Explicit semantic axes are now projected onto the per-target `ProductIR` before route selection, which prevents serialized target prose from reclassifying terminal spread baskets and lets T102/T126 bind their independent Stulz, Kirk, Monte Carlo, and Hurd-Zhou lanes. T102 now proves its Monte Carlo variant through authored `n_paths`, `n_steps`, `seed`, and `mc_method` spec overrides while its Stulz reference binds the raw analytical kernel. Other variant execution still must be proven by spec overrides or a canonical full-contract executable declaration; sparse legacy targets still rely on visibly inferred contracts, and `T13` remains an exact-bucket semantic-guard canary whose cached analytical artifact cannot honestly represent both theta-PDE variants and the analytical reference | Coherence can select and prove reusable numerical composition when it exists, but it deliberately does not invent missing methods, infer undeclared semantics for sparse legacy targets, or let one cached artifact impersonate several variants | `trellis/agent/comparison_target_contracts.py`, `trellis/agent/assembly_tools.py`, `trellis/agent/task_runtime.py`, `trellis/agent/executor.py`, `TASKS_PROOF_LEGACY.yaml`, `docs/developer/task_and_eval_loops.rst`, `docs/quant/pricing_stack.rst` | | L64 | **Legacy proof-task contracts remain broadly incomplete** — the task-manifest gate now validates all modern corpus envelopes strictly and freezes the legacy corpus's exact field-level debt and normalized task content behind a checked baseline. After the authored T02, T17, T102, and E22 repairs, that baseline contains 586 exact issue identities across 121 incomplete retained rows; those rows still lack one or more authored descriptions, economic contracts, market contracts, acceptance criteria, or explicit execution/hold dispositions. Product-specific field sufficiency remains owned by the semantic validators and repair tickets | New or worsened structural manifest debt fails the corpus gate, and the main task runner plus specific-id rerunner reject selected incomplete legacy rows before default market construction or code generation. The exact authored T02, T17, T102, and E22 rows pass that boundary without weakening the remaining debt. The legacy baseline is only a migration guard; it must not be interpreted as evidence that title-only rows are priceable, that every specialized proof harness is governed by the main runner boundary, or that every structurally valid modern product contract is semantically complete | `TASKS_PROOF_LEGACY.yaml`, `TASKS_PROOF_LEGACY_BASELINE.yaml`, `trellis/agent/task_manifest_validation.py`, `scripts/validate_task_manifests.py`, `scripts/run_tasks.py`, `scripts/rerun_ids.py`, `doc/plan/active__task-manifest-integrity.md` | -| L68 | **E22 cap-strip Monte Carlo is a forward-marginal proof** — the authored USD cap has twenty quarterly unadjusted periods, explicit fixing and payment dates, ACT/360 accrual, ACT/365 option time, ACT/ACT ISDA curve time, named OIS/SOFR-3M/Black-vol inputs, and seeded antithetic sampling. Both lanes value the same sum of individual caplet expectations | The 0.5% analytical-reference comparison supports this bounded proof. The Monte Carlo helper does not simulate a short-rate path or a joint forward-rate process, and its legacy n_steps, mean_reversion and sigma arguments are ignored; E22 cannot declare those controls. The proof does not establish calibration, shifted/normal volatility, seasoned fixing or production holiday support | `TASKS_PROOF_LEGACY.yaml`, `trellis/models/rate_cap_floor.py`, `trellis/agent/task_manifest_validation.py`, `trellis/agent/executor.py`, `tests/test_tasks/test_e22_cap_strip.py`, `docs/quant/pricing_stack.rst` | +| L68 | **E22 cap-strip Monte Carlo is a forward-marginal proof** — the authored USD cap has twenty quarterly unadjusted periods, explicit fixing and payment dates, ACT/360 accrual, ACT/365 option time, ACT/ACT ISDA curve time, named OIS/SOFR-3M/Black-vol inputs, and seeded antithetic sampling. Both lanes value the same sum of individual caplet expectations | The 0.5% analytical-reference comparison supports this bounded proof. The Monte Carlo helper and its typed lowering use independent forward marginals, not a short-rate path or a joint forward-rate process. Its legacy n_steps, mean_reversion and sigma arguments are ignored; E22 cannot declare those controls. The proof does not establish calibration, shifted/normal volatility, seasoned fixing or production holiday support | `TASKS_PROOF_LEGACY.yaml`, `trellis/models/rate_cap_floor.py`, `trellis/agent/task_manifest_validation.py`, `trellis/agent/executor.py`, `trellis/agent/family_lowering_ir.py`, `tests/test_tasks/test_e22_cap_strip.py`, `docs/quant/pricing_stack.rst` | | L65 | **Callable-bond coupons are scalar fixed-rate only** — the checked callable-bond cashflow, compiler, lattice, and PDE routes accept one scalar coupon rate rather than a dated variable-coupon schedule. Legacy task T09 asks for a step-up callable bond but does not author the coupon rates or effective dates, so its validated task contract now fails closed with an exact `variable_coupon_schedule` blocker and zero build attempts instead of using the old title-derived flat 5% fixture | Trellis can price the bounded fixed-coupon callable-bond cohort, but it cannot reasonably price step-up, step-down, floating, or otherwise variable-coupon callable bonds. T09 remains an expected honest block until QUA-1251 adds a reusable dated coupon primitive and an explicit schedule | `trellis/models/short_rate_fixed_income.py`, `trellis/instruments/callable_bond.py`, `trellis/models/callable_bond_pde.py`, `trellis/execution/compiler.py`, `trellis/agent/task_runtime.py`, `TASKS_PROOF_LEGACY.yaml`, `tests/test_agent/test_task_runtime.py` | | L66 | **Physical Bermudan swaption lattice support is a bounded one-factor, static-basis composition** — the strict route preserves explicit co-terminal swap tails, separate named discount/forecast curves, complete supported leg conventions, a provenance-complete named constant-parameter Hull-White set, and authored uniform-grid controls. It supports physical settlement, simple floating coupons with a deterministic additive forward basis, and ACT/365F model time. Stochastic basis, reset/payment convexity, compounded overnight coupons, amortizing or scheduled notionals, seasoned or pre-started fixed tails requiring accrued-settlement treatment, ACT/ACT ICMA coupon accrual without explicit quasi-coupon reference periods, parameterized day-of-month rolls, non-shipped calendar aliases, cash/annuity settlement, term-structured model volatility, multi-factor rates, native Greeks, and production convergence/error governance remain unsupported | Trellis can reasonably price the checked bounded physical dual-curve contract only when each adjusted first fixed accrual start is on or after exercise; it must fail closed rather than value a whole already-started fixed coupon, reinterpret richer Bermudan swaptions through the legacy T04 helper, use a European/Black fallback, or approximate a convention mapping. P005 is the exact executable evidence: its strict lattice lane prices and remains visible even though the paired Monte Carlo lane honestly blocks | `trellis/models/rate_swap_tail.py`, `trellis/models/hull_white_parameters.py`, `trellis/agent/semantic_contracts.py`, `trellis/agent/executor.py`, `trellis/agent/knowledge/canonical/routes.yaml`, `TASKS_EXTENSION.yaml`, `tests/test_models/test_rate_swap_tail.py`, `tests/test_agent/test_physical_bermudan_swaption_semantics.py`, `tests/test_tasks/test_p005_physical_bermudan_swaption.py`, `docs/quant/lattice_algebra.rst`, `docs/user_guide/pricing.rst` | | L58 | **FpML normalization is bounded to fixed-float IRS, physical European swaption, and scheduled cap/floor cohorts** — Support-contract version 1.0.0 distinguishes secure inspection, economic normalization, executable structural lowering, and paired conformance. `make_fpml_request(...)` and `trellis.io.fpml` securely inspect inline UTF-8 FpML 5.13 confirmation `dataDocument` payloads and normalize one regular, single-currency, constant-notional fixed-float swap into `StaticLegContractIR`, one physically settled European payer/receiver swaption into `ContractIR` with the complete swap nested under `underlying_contract`, or one regular single-currency constant-strike cap/floor into the existing signed `PeriodRateOptionStripLeg`. Swaptions reuse the structural resolved Black-76 declaration; cap/floors reuse the existing static strip declaration; historical settled premiums are reported separately and excluded from contract identity. `TASKS_FPML_CONFORMANCE.yaml` pairs all three admitted cohorts with independently specified native contracts and proves identity, projection, structural selection, market binding, price, and non-economic envelope invariance; its negative cohort certifies exact honest blockers with zero agent calls. This evidence does not widen support. Trellis still does not perform complete XSD validation, support other views/versions, resolve external references, bind imported seasoned coupons to historical fixing histories, classify vendor extension children, or normalize amortizing, compounding, stubbed, end-of-month or clamped high-day, cross-currency, OIS, inflation, lifecycle, package, cash-settled/Bermudan/American/partial/automatic/straddle swaption, unsettled-premium, cap/floor collar, stepped-strike, averaged, geared/spread, early-terminable, or other product forms | An admitted swap, physical European swaption, or scheduled cap/floor strip can price deterministically through shared structural execution when the caller declares a valuation party and valuation date and all cohort constraints hold. Unclassified extension children and all other FpML economics remain fail-closed with exact import, clarification, conflict, or unsupported-feature blockers; this is not general FpML pricing coverage | `TASKS_FPML_CONFORMANCE.yaml`, `trellis/io/fpml/`, `trellis/agent/fpml_conformance.py`, `trellis/agent/contract_ir.py`, `trellis/agent/static_leg_contract.py`, `trellis/agent/imported_documents.py`, `trellis/agent/platform_requests.py`, `trellis/platform/executor.py`, `docs/developer/fpml_support_matrix.rst`, `docs/developer/fpml_import.rst`, `docs/quant/contract_ir.rst`, `docs/quant/static_leg_contract_ir.rst`, `doc/plan/draft__fpml-interoperability-roadmap.md` | diff --git a/docs/developer/implementation_journey_prompt_to_price.md b/docs/developer/implementation_journey_prompt_to_price.md index bb3f0147..0d5e3af0 100644 --- a/docs/developer/implementation_journey_prompt_to_price.md +++ b/docs/developer/implementation_journey_prompt_to_price.md @@ -189,7 +189,17 @@ ACT/365 option time, ACT/ACT ISDA curve time, unadjusted dates, named market sce semantic bridge preserves these fields when specializing to analytical or Monte Carlo. Generated spec defaults and benchmark overrides carry the same dates and controls; optional strip terms keep their declared defaults -so smoke validation cannot invent a callable feature or collar strike. +so smoke validation cannot invent a callable feature or collar strike. A plain +unstructured cap/floor smoke fixture retains a usable common strike; an +authored strike uses its hydrated default. The rendered build request includes +the authored valuation measure, output unit and output currency. + +The cap/floor Monte Carlo family IR names scalar forward-rate marginals, +exact lognormal antithetic sampling and independent fixing observations under +each period's payment-forward measure. It does not describe that helper as +Hull-White, an OU transition, or a replayed short-rate path. Both the stress and +binding-first proof inventories use E22's `analytical` / `monte_carlo` targets +and the `analytical` reference. The exact validator accepts the authored source row and its canonical materialized market envelope. It rejects altered economics, controls or diff --git a/docs/quant/pricing_stack.rst b/docs/quant/pricing_stack.rst index 30b03fa7..bf53bd99 100644 --- a/docs/quant/pricing_stack.rst +++ b/docs/quant/pricing_stack.rst @@ -1300,7 +1300,10 @@ expiry, accrual, payment discount and Black volatility as the analytical leg. The 0.5% tolerance is a relative error in holder present value. This proof does not simulate a short-rate path or assert a joint model for caplet forwards; cross-caplet dependence is unnecessary for this linear -sum of individual option expectations. It does not validate calibration, +sum of individual option expectations. The compiler's Monte Carlo family IR +records scalar forward rates, exact lognormal antithetic sampling, and +independent period-payment-forward expectations, not Hull-White/OU dynamics. +It does not validate calibration, shifted/normal volatility, seasoned fixings or production holiday rules. Below those public callable wrappers, the reusable coupon/event/control layer diff --git a/docs/user_guide/pricing.rst b/docs/user_guide/pricing.rst index fcd710c8..bf62b66b 100644 --- a/docs/user_guide/pricing.rst +++ b/docs/user_guide/pricing.rst @@ -776,7 +776,9 @@ all quarterly accrual/fixing/payment dates, date conventions, named forecast and discount curves, Black volatility, sample count and random seed. Its analytical reference is the discounted Black caplet sum; its Monte Carlo comparison samples the same lognormal caplet forwards and allows -0.5% relative price error. This is a forward-marginal pricing proof, so it +0.5% relative price error. Both outputs are USD holder present values in +currency amounts, and those output terms are included in the build request. +This is a forward-marginal pricing proof, so it does not establish short-rate path simulation or general cap-market calibration support. The manifest fails admission if its required terms or controls are missing or changed. diff --git a/tests/evals/binding_first_exotic_proof.yaml b/tests/evals/binding_first_exotic_proof.yaml index 5b6941de..f2832476 100644 --- a/tests/evals/binding_first_exotic_proof.yaml +++ b/tests/evals/binding_first_exotic_proof.yaml @@ -40,9 +40,9 @@ E22: - forward_curve - black_vol_surface comparison_targets: - - mc_rate_cap - - black76_cap - reference_target: black76_cap + - analytical + - monte_carlo + reference_target: analytical requires_binding_ids: true forbidden_failure_patterns: - MissingCapabilityError diff --git a/tests/evals/stress_tasks.yaml b/tests/evals/stress_tasks.yaml index d3dbfcfa..e51e6919 100644 --- a/tests/evals/stress_tasks.yaml +++ b/tests/evals/stress_tasks.yaml @@ -23,9 +23,9 @@ E22: - forward_curve - black_vol_surface comparison_targets: - - mc_rate_cap - - black76_cap - reference_target: black76_cap + - analytical + - monte_carlo + reference_target: analytical forbidden_failure_patterns: - MissingCapabilityError diff --git a/tests/test_agent/test_family_lowering_ir.py b/tests/test_agent/test_family_lowering_ir.py index 4b4ba09d..951ba68b 100644 --- a/tests/test_agent/test_family_lowering_ir.py +++ b/tests/test_agent/test_family_lowering_ir.py @@ -294,13 +294,14 @@ def test_rate_style_swaption_monte_carlo_compiles_to_event_aware_family_ir(): ) -def test_rate_cap_floor_strip_monte_carlo_compiles_to_event_aware_family_ir(): +@pytest.mark.parametrize("instrument_class", ["cap", "floor"]) +def test_rate_cap_floor_strip_monte_carlo_compiles_to_event_aware_family_ir(instrument_class): from trellis.agent.semantic_contract_compiler import compile_semantic_contract from trellis.agent.semantic_contracts import make_period_rate_option_strip_contract contract = make_period_rate_option_strip_contract( - description="Black floorlet strip vs Hull-White Monte Carlo", - instrument_class="floor", + description="Uninformative rate strip label", + instrument_class=instrument_class, observation_schedule=("floor_schedule_placeholder",), preferred_method="monte_carlo", ) @@ -309,12 +310,18 @@ def test_rate_cap_floor_strip_monte_carlo_compiles_to_event_aware_family_ir(): family_ir = blueprint.dsl_lowering.family_ir assert isinstance(family_ir, EventAwareMonteCarloIR) assert family_ir.route_id == "monte_carlo_paths" - assert family_ir.product_instrument == "floor" + assert family_ir.product_instrument == instrument_class assert family_ir.payoff_family == "period_rate_option_strip" - assert family_ir.state_spec.state_variable == "short_rate" - assert family_ir.process_spec.process_family == "hull_white_1f" + assert family_ir.state_spec.state_variable == "forward_rate" + assert family_ir.process_spec.process_family == "independent_lognormal_forward_marginals" + assert family_ir.process_spec.simulation_scheme == "exact_lognormal" + assert family_ir.process_spec.process_tags == ("antithetic", "no_joint_forward_process") assert family_ir.helper_symbol == "price_rate_cap_floor_strip_monte_carlo" - assert family_ir.path_requirement_spec.requirement_kind == "event_replay" + assert family_ir.path_requirement_spec.requirement_kind == "independent_fixing_marginals" + assert family_ir.path_requirement_spec.replay_mode == "independent_periods" + assert family_ir.path_requirement_spec.stored_fields == ("forward_rate",) + assert family_ir.measure_spec.measure_family == "period_payment_forward" + assert family_ir.measure_spec.numeraire_binding == "payment_date_discount_factor" assert family_ir.payoff_reducer_spec.reducer_kind == "period_option_cashflow_strip" assert family_ir.market_mapping == "discount_curve_forward_curve_black_vol_to_rate_option_strip_mc" diff --git a/tests/test_tasks/test_e22_cap_strip.py b/tests/test_tasks/test_e22_cap_strip.py index bd02356c..ee563d3a 100644 --- a/tests/test_tasks/test_e22_cap_strip.py +++ b/tests/test_tasks/test_e22_cap_strip.py @@ -37,6 +37,27 @@ def test_e22_authored_contract_is_admitted_and_title_independent(): ) +@pytest.mark.parametrize( + "loader_name", + ["load_stress_task_manifest", "load_binding_first_exotic_proof_manifest"], +) +def test_e22_eval_manifests_keep_authored_target_and_reference_names(loader_name): + from trellis.agent import evals + + expectation = getattr(evals, loader_name)()["E22"] + assert expectation["comparison_targets"] == ["analytical", "monte_carlo"] + assert expectation["reference_target"] == "analytical" + + +def test_e22_structured_description_carries_authored_output_terms(): + from trellis.agent.task_runtime import task_to_description + + task = _task() + description = task_to_description(task) + for field in ("valuation_measure", "output_unit", "output_currency"): + assert f"{field}: {task['benchmark_contract'][field]}." in description + + def test_e22_structured_bridge_and_method_rebuild_preserve_the_contract(monkeypatch): from trellis.agent.semantic_contracts import specialize_semantic_contract_for_method from trellis.agent.task_runtime import ( @@ -88,18 +109,27 @@ def test_e22_generated_schema_keeps_explicit_schedule_and_controls(): compile(generated, "", "exec") -def test_cap_strip_smoke_fixture_does_not_invent_callable_or_collar_terms(): +@pytest.mark.parametrize("instrument_class", ["cap", "floor"]) +def test_cap_strip_smoke_fixture_preserves_plain_cap_floor_terms(instrument_class): from trellis.agent.executor import _generate_skeleton, _make_test_payoff from trellis.agent.planner import STATIC_SPECS + from trellis.agent.task_runtime import build_market_state_for_task + from trellis.models.rate_cap_floor import price_rate_cap_floor_strip_analytical schema = STATIC_SPECS["period_rate_option_strip"] namespace = {} - exec(_generate_skeleton(schema, "Non-callable cap"), namespace) + exec(_generate_skeleton(schema, f"Plain {instrument_class}"), namespace) payoff = _make_test_payoff(namespace[schema.class_name], schema, date(2024, 11, 15)) assert payoff._spec.call_price is None assert payoff._spec.exercise_dates is None assert payoff._spec.cap_strike is None assert payoff._spec.floor_strike is None + market, _ = build_market_state_for_task(_task()) + price = price_rate_cap_floor_strip_analytical( + market, payoff._spec, instrument_class=instrument_class + ) + assert payoff._spec.strike == 0.05 + assert price > 0.0 def test_e22_admits_materialized_market_and_rejects_market_override(): @@ -311,6 +341,16 @@ def test_e22_loaded_contract_prices_both_declared_lanes_end_to_end( task = {**_task(), "title": "Uninformative label"} observed = [] + compiled_metadata = {} + materialize = executor._materialize_deterministic_exact_binding_module + + def observed_materialize(skeleton, generation_plan, **kwargs): + target = kwargs.get("comparison_target") + if target: + compiled_metadata[target] = platform_requests._semantic_blueprint_summary( + kwargs["semantic_blueprint"] + ) + return materialize(skeleton, generation_plan, **kwargs) def observed_price(payoff, market): observed.append((payoff._spec, market.settlement)) @@ -326,6 +366,9 @@ def write_generated_module(module_path: str, code: str) -> Path: monkeypatch.setattr(executor, "TRELLIS_PACKAGE_ROOT", tmp_path / "trellis") monkeypatch.setattr(executor, "_REPO_REVISION", "test") monkeypatch.setattr(executor, "write_module", write_generated_module) + monkeypatch.setattr( + executor, "_materialize_deterministic_exact_binding_module", observed_materialize + ) monkeypatch.setattr( analytical_traces, "TRACE_ROOT", tmp_path / "traces" / "analytical" ) @@ -363,6 +406,17 @@ def write_generated_module(module_path: str, code: str) -> Path: assert result["attempts"] == 0 assert result["token_usage_summary"]["call_count"] == 0 assert result["instrument_type"] == "cap" + assert set(compiled_metadata) == {"analytical", "monte_carlo"} + mc_ir = compiled_metadata["monte_carlo"]["dsl_family_ir"] + assert mc_ir["state_spec"]["state_variable"] == "forward_rate" + assert mc_ir["process_spec"] == { + "process_family": "independent_lognormal_forward_marginals", + "simulation_scheme": "exact_lognormal", + "process_tags": ["antithetic", "no_joint_forward_process"], + } + assert mc_ir["path_requirement_spec"]["requirement_kind"] == "independent_fixing_marginals" + assert mc_ir["measure_spec"]["measure_family"] == "period_payment_forward" + assert mc_ir["helper_symbol"] == "price_rate_cap_floor_strip_monte_carlo" assert result["runtime_contract"]["simulation_seed"] == 42 assert ( result["runtime_contract"]["simulation_identity"]["seed_source"] diff --git a/trellis/agent/benchmark_contracts.py b/trellis/agent/benchmark_contracts.py index a50da3f5..a21b566b 100644 --- a/trellis/agent/benchmark_contracts.py +++ b/trellis/agent/benchmark_contracts.py @@ -926,6 +926,7 @@ def _benchmark_detail_lines( lines.append(f"Day count: {contract['day_count']}.") for field in ( "model_time_day_count", "discount_curve_day_count", "forecast_curve_day_count", + "valuation_measure", "output_unit", "output_currency", "calendar_name", "business_day_adjustment", "fixing_rule", "payment_rule", "fixing_lag_days", "payment_lag_days", "mc_distribution", "sampling", "n_paths", "seed", diff --git a/trellis/agent/executor.py b/trellis/agent/executor.py index bb3fd800..552346fd 100644 --- a/trellis/agent/executor.py +++ b/trellis/agent/executor.py @@ -3843,6 +3843,11 @@ def _make_test_payoff( # values would turn a plain cap into a different, unsupported product. for field in spec_schema.fields: if field.default is not None: + # The shared schema makes strike optional for collars, but a + # plain unstructured cap/floor still needs its smoke strike. + # Authored strikes keep their hydrated dataclass default. + if field.name == "strike" and field.default == "None": + continue name_defaults.pop(field.name, None) description = getattr(payoff_cls, "__doc__", "") or getattr(module, "__doc__", "") or "" diff --git a/trellis/agent/family_lowering_ir.py b/trellis/agent/family_lowering_ir.py index 56f61886..e6acd6d9 100644 --- a/trellis/agent/family_lowering_ir.py +++ b/trellis/agent/family_lowering_ir.py @@ -853,7 +853,7 @@ class ExerciseLatticeProfile: _MC_MARKET_MAPPINGS = { ("hull_white_1f", "swaption"): "discount_curve_forward_curve_black_vol_to_short_rate_mc", - ("hull_white_1f", "period_rate_option_strip"): "discount_curve_forward_curve_black_vol_to_rate_option_strip_mc", + ("independent_lognormal_forward_marginals", "period_rate_option_strip"): "discount_curve_forward_curve_black_vol_to_rate_option_strip_mc", ("heston", "vanilla_option"): "heston_model_parameters_to_stochastic_vol_mc", ("local_vol_1d", ""): "equity_spot_discount_local_vol_to_mc", ("gbm_1d", ""): "equity_spot_discount_black_vol_to_mc", @@ -862,7 +862,7 @@ class ExerciseLatticeProfile: _MC_HELPER_BINDINGS = { ("heston", "vanilla_option", "european"): "price_heston_option_monte_carlo", - ("hull_white_1f", "period_rate_option_strip", ""): "price_rate_cap_floor_strip_monte_carlo", + ("independent_lognormal_forward_marginals", "period_rate_option_strip", ""): "price_rate_cap_floor_strip_monte_carlo", } @@ -1950,8 +1950,14 @@ def _build_event_aware_monte_carlo_ir( payoff_reducer_spec=payoff_reducer_spec, control_spec=control_spec, measure_spec=MCMeasureSpec( - measure_family="risk_neutral", - numeraire_binding="discount_curve", + measure_family=( + "period_payment_forward" if _mc_uses_forward_marginal_strip(product) + else "risk_neutral" + ), + numeraire_binding=( + "payment_date_discount_factor" if _mc_uses_forward_marginal_strip(product) + else "discount_curve" + ), ), event_timeline=event_timeline, event_specs=tuple( @@ -1965,6 +1971,16 @@ def _build_event_aware_monte_carlo_ir( ) +def _mc_uses_forward_marginal_strip(product) -> bool: + """Identify the cap/floor helper, which samples no short-rate or joint path.""" + return ( + getattr(product, "semantic_id", "") == "period_rate_option_strip" + and getattr(product, "payoff_family", "") == "period_rate_option_strip" + and getattr(product, "instrument_class", "") in {"cap", "floor"} + and getattr(product, "exercise_style", "") == "none" + ) + + def _mc_uses_terminal_only_contract(product) -> bool: """Return whether the product should lower onto a terminal-only MC contract.""" payoff_family = _canonical_semantic_family(getattr(product, "payoff_family", "")) @@ -1995,6 +2011,13 @@ def _mc_control_spec_for_product(product) -> MCControlSpec | None: def _mc_state_spec_for_product(product) -> MCStateSpec: """Infer the bounded Monte Carlo state contract from semantic metadata.""" + if _mc_uses_forward_marginal_strip(product): + return MCStateSpec( + state_variable="forward_rate", + dimension=1, + state_tags=("independent_fixing_marginals",), + state_layout="scalar", + ) model_family = str(getattr(product, "model_family", "") or "").strip().lower() state_tags = _tuple_unique(("terminal_markov", *_state_tags(product))) if model_family in {"interest_rate", "short_rate"}: @@ -2028,6 +2051,12 @@ def _mc_process_spec_for_product( route_id: str, ) -> MCProcessSpec: """Infer the bounded Monte Carlo process contract from semantic metadata.""" + if _mc_uses_forward_marginal_strip(product): + return MCProcessSpec( + process_family="independent_lognormal_forward_marginals", + simulation_scheme="exact_lognormal", + process_tags=("antithetic", "no_joint_forward_process"), + ) if _binding_has_symbol(binding_spec, "state_process", "LocalVol") or _binding_has_symbol( binding_spec, "pricing_kernel", @@ -2244,6 +2273,14 @@ def _mc_path_requirement_spec_for_product( event_timeline: tuple[MCEventTimeSpec, ...], ) -> MCPathRequirementSpec: """Infer the reduced-state contract needed by the bounded MC family.""" + if _mc_uses_forward_marginal_strip(product): + return MCPathRequirementSpec( + requirement_kind="independent_fixing_marginals", + snapshot_schedule_role="observation_dates", + reducer_kinds=_mc_reducer_kinds_for_product(product), + replay_mode="independent_periods", + stored_fields=("forward_rate",), + ) if event_timeline: requirement_kind = "event_replay" replay_mode = "deterministic_timeline" From 24e7f5e996544bb97900c99c0fab0c48d74cf9c3 Mon Sep 17 00:00:00 2001 From: steveya Date: Sun, 13 Sep 2026 02:35:05 -0400 Subject: [PATCH 3/3] Admit bounded cap-strip forward-marginal Monte Carlo routes --- LIMITATIONS.md | 2 +- .../implementation_journey_prompt_to_price.md | 8 ++ docs/quant/pricing_stack.rst | 5 + .../test_cap_strip_route_admission.py | 107 ++++++++++++++++++ tests/test_agent/test_route_registry.py | 6 +- tests/test_tasks/test_e22_cap_strip.py | 19 +++- trellis/agent/knowledge/canonical/routes.yaml | 7 +- trellis/agent/route_registry.py | 78 +++++++++++++ 8 files changed, 226 insertions(+), 6 deletions(-) create mode 100644 tests/test_agent/test_cap_strip_route_admission.py diff --git a/LIMITATIONS.md b/LIMITATIONS.md index 163e6d53..f743d639 100644 --- a/LIMITATIONS.md +++ b/LIMITATIONS.md @@ -70,7 +70,7 @@ ground truth until revalidated. | L49 | **Agent-cycle and institutional approval evidence are governed internal review surfaces, not external model certification** — Trellis now exposes a stable `agent_cycle` result surface for quant/critic/arbiter/model-validator evidence, model-promotion eligibility, and benchmark trigger rates, and `policy_bundle.production.institutional` can require approval, model-review, snapshot, run-artifact, and audit-bundle evidence before production execution, but these surfaces only certify recorded internal governance evidence for the run or model version | Product and desk review surfaces can show why a cycle passed, failed, was unavailable, or lacked required institutional approval artifacts, but they must not be read as external model approval, regulatory sign-off, xVA/FpML coverage, or correctness beyond the recorded validation scope | `trellis/agent/cycle_surface.py`, `trellis/agent/task_runtime.py`, `trellis/platform/models.py`, `trellis/platform/policies.py`, `trellis/platform/services/pricing_service.py`, `docs/developer/audit_and_observability.rst`, `docs/developer/hosting_and_configuration.rst`, `docs/user_guide/pricing.rst` | | L57 | **Typed comparison-target coherence exposes rather than fills missing numerical composition** — Explicit targets carry canonical method, route/binding, variant, validation, and semantic identities, and the runtime rejects partial explicit target sets, ambiguous references, missing declarations, and unbound shared artifacts before pricing. Explicit semantic axes are now projected onto the per-target `ProductIR` before route selection, which prevents serialized target prose from reclassifying terminal spread baskets and lets T102/T126 bind their independent Stulz, Kirk, Monte Carlo, and Hurd-Zhou lanes. T102 now proves its Monte Carlo variant through authored `n_paths`, `n_steps`, `seed`, and `mc_method` spec overrides while its Stulz reference binds the raw analytical kernel. Other variant execution still must be proven by spec overrides or a canonical full-contract executable declaration; sparse legacy targets still rely on visibly inferred contracts, and `T13` remains an exact-bucket semantic-guard canary whose cached analytical artifact cannot honestly represent both theta-PDE variants and the analytical reference | Coherence can select and prove reusable numerical composition when it exists, but it deliberately does not invent missing methods, infer undeclared semantics for sparse legacy targets, or let one cached artifact impersonate several variants | `trellis/agent/comparison_target_contracts.py`, `trellis/agent/assembly_tools.py`, `trellis/agent/task_runtime.py`, `trellis/agent/executor.py`, `TASKS_PROOF_LEGACY.yaml`, `docs/developer/task_and_eval_loops.rst`, `docs/quant/pricing_stack.rst` | | L64 | **Legacy proof-task contracts remain broadly incomplete** — the task-manifest gate now validates all modern corpus envelopes strictly and freezes the legacy corpus's exact field-level debt and normalized task content behind a checked baseline. After the authored T02, T17, T102, and E22 repairs, that baseline contains 586 exact issue identities across 121 incomplete retained rows; those rows still lack one or more authored descriptions, economic contracts, market contracts, acceptance criteria, or explicit execution/hold dispositions. Product-specific field sufficiency remains owned by the semantic validators and repair tickets | New or worsened structural manifest debt fails the corpus gate, and the main task runner plus specific-id rerunner reject selected incomplete legacy rows before default market construction or code generation. The exact authored T02, T17, T102, and E22 rows pass that boundary without weakening the remaining debt. The legacy baseline is only a migration guard; it must not be interpreted as evidence that title-only rows are priceable, that every specialized proof harness is governed by the main runner boundary, or that every structurally valid modern product contract is semantically complete | `TASKS_PROOF_LEGACY.yaml`, `TASKS_PROOF_LEGACY_BASELINE.yaml`, `trellis/agent/task_manifest_validation.py`, `scripts/validate_task_manifests.py`, `scripts/run_tasks.py`, `scripts/rerun_ids.py`, `doc/plan/active__task-manifest-integrity.md` | -| L68 | **E22 cap-strip Monte Carlo is a forward-marginal proof** — the authored USD cap has twenty quarterly unadjusted periods, explicit fixing and payment dates, ACT/360 accrual, ACT/365 option time, ACT/ACT ISDA curve time, named OIS/SOFR-3M/Black-vol inputs, and seeded antithetic sampling. Both lanes value the same sum of individual caplet expectations | The 0.5% analytical-reference comparison supports this bounded proof. The Monte Carlo helper and its typed lowering use independent forward marginals, not a short-rate path or a joint forward-rate process. Its legacy n_steps, mean_reversion and sigma arguments are ignored; E22 cannot declare those controls. The proof does not establish calibration, shifted/normal volatility, seasoned fixing or production holiday support | `TASKS_PROOF_LEGACY.yaml`, `trellis/models/rate_cap_floor.py`, `trellis/agent/task_manifest_validation.py`, `trellis/agent/executor.py`, `trellis/agent/family_lowering_ir.py`, `tests/test_tasks/test_e22_cap_strip.py`, `docs/quant/pricing_stack.rst` | +| L68 | **E22 cap-strip Monte Carlo is a forward-marginal proof** — the authored USD cap has twenty quarterly unadjusted periods, explicit fixing and payment dates, ACT/360 accrual, ACT/365 option time, ACT/ACT ISDA curve time, named OIS/SOFR-3M/Black-vol inputs, and seeded antithetic sampling. Both lanes value the same sum of individual caplet expectations | The 0.5% analytical-reference comparison supports this bounded proof. The Monte Carlo helper and its typed lowering use independent forward marginals, not a short-rate path or a joint forward-rate process. Ordinary route admission restricts these capabilities to the existing Black cap/floor helper profile and blocks incompatible product/state/process/path/reducer/measure or calibration combinations. Its legacy n_steps, mean_reversion and sigma arguments are ignored; E22 cannot declare those controls. The proof does not establish calibration, shifted/normal volatility, seasoned fixing or production holiday support | `TASKS_PROOF_LEGACY.yaml`, `trellis/models/rate_cap_floor.py`, `trellis/agent/task_manifest_validation.py`, `trellis/agent/executor.py`, `trellis/agent/family_lowering_ir.py`, `trellis/agent/route_registry.py`, `tests/test_tasks/test_e22_cap_strip.py`, `docs/quant/pricing_stack.rst` | | L65 | **Callable-bond coupons are scalar fixed-rate only** — the checked callable-bond cashflow, compiler, lattice, and PDE routes accept one scalar coupon rate rather than a dated variable-coupon schedule. Legacy task T09 asks for a step-up callable bond but does not author the coupon rates or effective dates, so its validated task contract now fails closed with an exact `variable_coupon_schedule` blocker and zero build attempts instead of using the old title-derived flat 5% fixture | Trellis can price the bounded fixed-coupon callable-bond cohort, but it cannot reasonably price step-up, step-down, floating, or otherwise variable-coupon callable bonds. T09 remains an expected honest block until QUA-1251 adds a reusable dated coupon primitive and an explicit schedule | `trellis/models/short_rate_fixed_income.py`, `trellis/instruments/callable_bond.py`, `trellis/models/callable_bond_pde.py`, `trellis/execution/compiler.py`, `trellis/agent/task_runtime.py`, `TASKS_PROOF_LEGACY.yaml`, `tests/test_agent/test_task_runtime.py` | | L66 | **Physical Bermudan swaption lattice support is a bounded one-factor, static-basis composition** — the strict route preserves explicit co-terminal swap tails, separate named discount/forecast curves, complete supported leg conventions, a provenance-complete named constant-parameter Hull-White set, and authored uniform-grid controls. It supports physical settlement, simple floating coupons with a deterministic additive forward basis, and ACT/365F model time. Stochastic basis, reset/payment convexity, compounded overnight coupons, amortizing or scheduled notionals, seasoned or pre-started fixed tails requiring accrued-settlement treatment, ACT/ACT ICMA coupon accrual without explicit quasi-coupon reference periods, parameterized day-of-month rolls, non-shipped calendar aliases, cash/annuity settlement, term-structured model volatility, multi-factor rates, native Greeks, and production convergence/error governance remain unsupported | Trellis can reasonably price the checked bounded physical dual-curve contract only when each adjusted first fixed accrual start is on or after exercise; it must fail closed rather than value a whole already-started fixed coupon, reinterpret richer Bermudan swaptions through the legacy T04 helper, use a European/Black fallback, or approximate a convention mapping. P005 is the exact executable evidence: its strict lattice lane prices and remains visible even though the paired Monte Carlo lane honestly blocks | `trellis/models/rate_swap_tail.py`, `trellis/models/hull_white_parameters.py`, `trellis/agent/semantic_contracts.py`, `trellis/agent/executor.py`, `trellis/agent/knowledge/canonical/routes.yaml`, `TASKS_EXTENSION.yaml`, `tests/test_models/test_rate_swap_tail.py`, `tests/test_agent/test_physical_bermudan_swaption_semantics.py`, `tests/test_tasks/test_p005_physical_bermudan_swaption.py`, `docs/quant/lattice_algebra.rst`, `docs/user_guide/pricing.rst` | | L58 | **FpML normalization is bounded to fixed-float IRS, physical European swaption, and scheduled cap/floor cohorts** — Support-contract version 1.0.0 distinguishes secure inspection, economic normalization, executable structural lowering, and paired conformance. `make_fpml_request(...)` and `trellis.io.fpml` securely inspect inline UTF-8 FpML 5.13 confirmation `dataDocument` payloads and normalize one regular, single-currency, constant-notional fixed-float swap into `StaticLegContractIR`, one physically settled European payer/receiver swaption into `ContractIR` with the complete swap nested under `underlying_contract`, or one regular single-currency constant-strike cap/floor into the existing signed `PeriodRateOptionStripLeg`. Swaptions reuse the structural resolved Black-76 declaration; cap/floors reuse the existing static strip declaration; historical settled premiums are reported separately and excluded from contract identity. `TASKS_FPML_CONFORMANCE.yaml` pairs all three admitted cohorts with independently specified native contracts and proves identity, projection, structural selection, market binding, price, and non-economic envelope invariance; its negative cohort certifies exact honest blockers with zero agent calls. This evidence does not widen support. Trellis still does not perform complete XSD validation, support other views/versions, resolve external references, bind imported seasoned coupons to historical fixing histories, classify vendor extension children, or normalize amortizing, compounding, stubbed, end-of-month or clamped high-day, cross-currency, OIS, inflation, lifecycle, package, cash-settled/Bermudan/American/partial/automatic/straddle swaption, unsettled-premium, cap/floor collar, stepped-strike, averaged, geared/spread, early-terminable, or other product forms | An admitted swap, physical European swaption, or scheduled cap/floor strip can price deterministically through shared structural execution when the caller declares a valuation party and valuation date and all cohort constraints hold. Unclassified extension children and all other FpML economics remain fail-closed with exact import, clarification, conflict, or unsupported-feature blockers; this is not general FpML pricing coverage | `TASKS_FPML_CONFORMANCE.yaml`, `trellis/io/fpml/`, `trellis/agent/fpml_conformance.py`, `trellis/agent/contract_ir.py`, `trellis/agent/static_leg_contract.py`, `trellis/agent/imported_documents.py`, `trellis/agent/platform_requests.py`, `trellis/platform/executor.py`, `docs/developer/fpml_support_matrix.rst`, `docs/developer/fpml_import.rst`, `docs/quant/contract_ir.rst`, `docs/quant/static_leg_contract_ir.rst`, `doc/plan/draft__fpml-interoperability-roadmap.md` | diff --git a/docs/developer/implementation_journey_prompt_to_price.md b/docs/developer/implementation_journey_prompt_to_price.md index 0d5e3af0..c2d31949 100644 --- a/docs/developer/implementation_journey_prompt_to_price.md +++ b/docs/developer/implementation_journey_prompt_to_price.md @@ -201,6 +201,14 @@ Hull-White, an OU transition, or a replayed short-rate path. Both the stress and binding-first proof inventories use E22's `analytical` / `monte_carlo` targets and the `analytical` reference. +The canonical Monte Carlo route declares these forward-marginal capabilities, +with admission restricted to the existing Black cap/floor helper profile. +Mismatched product, state, process, path requirement, reducer, measure, helper +or calibration assumptions return `unsupported_forward_marginal_profile` +failures before ordinary generation. This is not generic joint-forward process +support. The loaded E22 end-to-end test also evaluates the ordinary pre-generation +gate, so an exact-binding replay cannot hide a route-admission failure. + The exact validator accepts the authored source row and its canonical materialized market envelope. It rejects altered economics, controls or scenario values before build. The title is descriptive only, and the diff --git a/docs/quant/pricing_stack.rst b/docs/quant/pricing_stack.rst index bf53bd99..905b458c 100644 --- a/docs/quant/pricing_stack.rst +++ b/docs/quant/pricing_stack.rst @@ -1303,6 +1303,11 @@ caplet forwards; cross-caplet dependence is unnecessary for this linear sum of individual option expectations. The compiler's Monte Carlo family IR records scalar forward rates, exact lognormal antithetic sampling, and independent period-payment-forward expectations, not Hull-White/OU dynamics. +Ordinary Monte Carlo route admission permits this existing cap/floor helper +profile only when the product, state, process, fixing-marginal storage, +cashflow reducer and payment-forward measure agree. Declaring these route +capabilities does not enable a general joint-forward model; incompatible +profiles and calibration requests block before generation. It does not validate calibration, shifted/normal volatility, seasoned fixings or production holiday rules. diff --git a/tests/test_agent/test_cap_strip_route_admission.py b/tests/test_agent/test_cap_strip_route_admission.py new file mode 100644 index 00000000..d7489edb --- /dev/null +++ b/tests/test_agent/test_cap_strip_route_admission.py @@ -0,0 +1,107 @@ +"""The existing caplet helper is admitted without implying a joint-forward model.""" + +from dataclasses import replace +from types import SimpleNamespace + +import pytest + +from trellis.agent.build_gate import evaluate_pre_generation_gate +from trellis.agent.route_registry import evaluate_route_admissibility, find_route_by_id +from trellis.agent.semantic_contract_compiler import compile_semantic_contract +from trellis.agent.semantic_contracts import make_period_rate_option_strip_contract + + +def _blueprint(instrument_class="cap"): + return compile_semantic_contract( + make_period_rate_option_strip_contract( + description="Uninformative label", + instrument_class=instrument_class, + observation_schedule=("2025-02-15", "2025-05-15"), + preferred_method="monte_carlo", + ), + preferred_method="monte_carlo", + ) + + +@pytest.mark.parametrize("instrument_class", ["cap", "floor"]) +def test_forward_marginal_cap_floor_passes_ordinary_route_and_generation_gate( + instrument_class, +): + blueprint = _blueprint(instrument_class) + route = find_route_by_id("monte_carlo_paths") + decision = evaluate_route_admissibility(route, semantic_blueprint=blueprint) + assert decision.ok, decision.failures + plan = SimpleNamespace(primitive_plan=SimpleNamespace(route=route.id)) + gate = evaluate_pre_generation_gate(None, plan, semantic_blueprint=blueprint) + assert gate.decision == "proceed", gate.reason + + +@pytest.mark.parametrize( + "section,field,value", + [ + ("product", "instrument_class", "swaption"), + ("product", "exercise_style", "bermudan"), + ("product", "model_family", "equity_diffusion"), + ("product", "payoff_family", "path_dependent_rate_option"), + ("product", "term_fields", {"mc_distribution": "joint_forward_process"}), + ("product", "term_fields", {"model": "normal"}), + ("product", "term_fields", {"sampling": "pseudo_random"}), + ("family", "product_instrument", "floor"), + ("family", "helper_symbol", "price_event_aware_monte_carlo"), + ("family", "calibration_binding", object()), + ("state_spec", "state_variable", "short_rate"), + ("state_spec", "dimension", 2), + ("state_spec", "state_tags", ("terminal_markov",)), + ("state_spec", "state_layout", "vector"), + ("process_spec", "process_family", "hull_white_1f"), + ("process_spec", "simulation_scheme", "exact_ou"), + ("process_spec", "process_tags", ()), + ("path_requirement_spec", "requirement_kind", "full_path"), + ("path_requirement_spec", "snapshot_schedule_role", "decision_dates"), + ("path_requirement_spec", "replay_mode", "deterministic_timeline"), + ("path_requirement_spec", "reducer_kinds", ("running_average",)), + ("path_requirement_spec", "stored_fields", ("short_rate",)), + ("payoff_reducer_spec", "reducer_kind", "compiled_schedule_payoff"), + ("payoff_reducer_spec", "output_semantics", "joint_forward_payoff"), + ("control_spec", "control_style", "issuer_call"), + ("measure_spec", "measure_family", "risk_neutral"), + ("measure_spec", "numeraire_binding", "discount_curve"), + ("blueprint", "calibration_step", object()), + ], +) +def test_forward_marginal_profile_mismatches_honestly_block_before_generation( + section, field, value +): + blueprint = _blueprint() + if section == "product": + blueprint = replace( + blueprint, + contract=replace( + blueprint.contract, + product=replace(blueprint.contract.product, **{field: value}), + ), + ) + elif section == "blueprint": + blueprint = replace(blueprint, **{field: value}) + else: + family = blueprint.dsl_lowering.family_ir + if section == "family": + family = replace(family, **{field: value}) + else: + family = replace( + family, **{section: replace(getattr(family, section), **{field: value})} + ) + blueprint = replace( + blueprint, dsl_lowering=replace(blueprint.dsl_lowering, family_ir=family) + ) + route = find_route_by_id("monte_carlo_paths") + decision = evaluate_route_admissibility(route, semantic_blueprint=blueprint) + assert not decision.ok + assert any( + failure.startswith("unsupported_forward_marginal_profile:") + for failure in decision.failures + ) + plan = SimpleNamespace(primitive_plan=SimpleNamespace(route=route.id)) + gate = evaluate_pre_generation_gate(None, plan, semantic_blueprint=blueprint) + assert gate.decision == "block" + assert gate.route_admissibility_failures == decision.failures diff --git a/tests/test_agent/test_route_registry.py b/tests/test_agent/test_route_registry.py index 8ebcff75..bc02f6f1 100644 --- a/tests/test_agent/test_route_registry.py +++ b/tests/test_agent/test_route_registry.py @@ -1372,12 +1372,15 @@ def test_admissibility_hydrates_process_and_path_contracts(self, registry): local_vol = find_route_by_id("local_vol_monte_carlo", registry) assert generic is not None - assert generic.admissibility.supported_process_families == ("gbm_1d", "hull_white_1f") + assert generic.admissibility.supported_process_families == ( + "gbm_1d", "hull_white_1f", "independent_lognormal_forward_marginals", + ) assert generic.admissibility.supported_state_tags == ( "pathwise_only", "terminal_markov", "recombining_safe", "schedule_state", + "independent_fixing_marginals", ) assert generic.admissibility.supported_path_requirement_kinds == ( "terminal_only", @@ -1385,6 +1388,7 @@ def test_admissibility_hydrates_process_and_path_contracts(self, registry): "event_snapshots", "event_replay", "reducer_state", + "independent_fixing_marginals", ) assert generic.admissibility.supports_calibration is True assert local_vol is not None diff --git a/tests/test_tasks/test_e22_cap_strip.py b/tests/test_tasks/test_e22_cap_strip.py index ee563d3a..facd7236 100644 --- a/tests/test_tasks/test_e22_cap_strip.py +++ b/tests/test_tasks/test_e22_cap_strip.py @@ -329,12 +329,14 @@ def test_e22_loaded_contract_prices_both_declared_lanes_end_to_end( """Defend the authored contract across compilation, hydration and comparison.""" import sys from pathlib import Path + from types import SimpleNamespace import trellis.agent.analytical_traces as analytical_traces import trellis.agent.executor as executor import trellis.agent.model_audit as model_audit import trellis.agent.platform_requests as platform_requests import trellis.agent.platform_traces as platform_traces + from trellis.agent.build_gate import evaluate_pre_generation_gate from trellis.agent.offline_agents import offline_local_agent_run_scope from trellis.agent.task_runtime import run_task from trellis.engine.payoff_pricer import price_payoff @@ -342,13 +344,24 @@ def test_e22_loaded_contract_prices_both_declared_lanes_end_to_end( task = {**_task(), "title": "Uninformative label"} observed = [] compiled_metadata = {} + ordinary_gate_decisions = {} materialize = executor._materialize_deterministic_exact_binding_module def observed_materialize(skeleton, generation_plan, **kwargs): target = kwargs.get("comparison_target") if target: + blueprint = kwargs["semantic_blueprint"] + route_id = blueprint.dsl_lowering.family_ir.route_id + if target == "monte_carlo": + assert route_id == "monte_carlo_paths" compiled_metadata[target] = platform_requests._semantic_blueprint_summary( - kwargs["semantic_blueprint"] + blueprint + ) + # Exact binding has no primitive plan. Exercise ordinary admission + # explicitly using the canonical route emitted by the real compiler. + ordinary_plan = SimpleNamespace(primitive_plan=SimpleNamespace(route=route_id)) + ordinary_gate_decisions[target] = evaluate_pre_generation_gate( + None, ordinary_plan, semantic_blueprint=blueprint ) return materialize(skeleton, generation_plan, **kwargs) @@ -407,6 +420,10 @@ def write_generated_module(module_path: str, code: str) -> Path: assert result["token_usage_summary"]["call_count"] == 0 assert result["instrument_type"] == "cap" assert set(compiled_metadata) == {"analytical", "monte_carlo"} + assert set(ordinary_gate_decisions) == {"analytical", "monte_carlo"} + for decision in ordinary_gate_decisions.values(): + assert decision.decision == "proceed", decision.reason + assert not decision.route_admissibility_failures mc_ir = compiled_metadata["monte_carlo"]["dsl_family_ir"] assert mc_ir["state_spec"]["state_variable"] == "forward_rate" assert mc_ir["process_spec"] == { diff --git a/trellis/agent/knowledge/canonical/routes.yaml b/trellis/agent/knowledge/canonical/routes.yaml index 4ce9cd02..b188b7ed 100644 --- a/trellis/agent/knowledge/canonical/routes.yaml +++ b/trellis/agent/knowledge/canonical/routes.yaml @@ -513,9 +513,10 @@ routes: multicurrency_support: single_currency_only supported_outputs: [price, scenario_pnl] supports_sensitivity_outputs: true - supported_state_tags: [pathwise_only, terminal_markov, recombining_safe, schedule_state] - supported_process_families: [gbm_1d, hull_white_1f] - supported_path_requirement_kinds: [terminal_only, full_path, event_snapshots, event_replay, reducer_state] + # Forward marginals are admitted only as the bounded cap/floor helper profile. + supported_state_tags: [pathwise_only, terminal_markov, recombining_safe, schedule_state, independent_fixing_marginals] + supported_process_families: [gbm_1d, hull_white_1f, independent_lognormal_forward_marginals] + supported_path_requirement_kinds: [terminal_only, full_path, event_snapshots, event_replay, reducer_state, independent_fixing_marginals] supports_calibration: true primitives: - module: trellis.models.processes.gbm diff --git a/trellis/agent/route_registry.py b/trellis/agent/route_registry.py index 9e06d445..e597f2cb 100644 --- a/trellis/agent/route_registry.py +++ b/trellis/agent/route_registry.py @@ -1875,6 +1875,7 @@ def evaluate_route_admissibility( ): failures.append(f"unsupported_characteristic_family:{characteristic_family}") if isinstance(family_ir, EventAwareMonteCarloIR): + failures.extend(_forward_marginal_profile_failures(product, family_ir, calibration_step)) supported_processes = set(admissibility.supported_process_families) process_family = str(getattr(getattr(family_ir, "process_spec", None), "process_family", "")).strip() if supported_processes and process_family and process_family not in supported_processes: @@ -1906,6 +1907,83 @@ def evaluate_route_admissibility( ) +def _forward_marginal_profile_failures(product, family_ir, calibration_step) -> tuple[str, ...]: + """Keep the cap/floor helper capability separate from general MC processes. + + The route catalog declares a union of capabilities, not permission to mix + independent fixing marginals with Hull-White state, joint path payoffs, or + calibration. This profile describes only the existing Black strip helper. + """ + process_family = "independent_lognormal_forward_marginals" + marginal_tag = "independent_fixing_marginals" + strip_family = "period_rate_option_strip" + helper = "price_rate_cap_floor_strip_monte_carlo" + if not ( + getattr(product, "payoff_family", "") == strip_family + or family_ir.payoff_family == strip_family + or family_ir.helper_symbol == helper + or getattr(family_ir.process_spec, "process_family", "") == process_family + or marginal_tag in (getattr(family_ir.state_spec, "state_tags", ()) or ()) + or getattr(family_ir.path_requirement_spec, "requirement_kind", "") == marginal_tag + ): + return () + + failures = [] + instrument = getattr(product, "instrument_class", "") + if ( + getattr(product, "semantic_id", "") != strip_family + or getattr(product, "payoff_family", "") != strip_family + or instrument not in {"cap", "floor"} + or getattr(product, "exercise_style", "") != "none" + or getattr(product, "model_family", "") not in {"interest_rate", "short_rate"} + or getattr(getattr(product, "controller_protocol", None), "controller_style", "") != "identity" + or family_ir.product_instrument != instrument + or family_ir.payoff_family != strip_family + or family_ir.helper_symbol != helper + ): + failures.append("unsupported_forward_marginal_profile:cap_floor_strip_helper") + + # Optional authored terms may narrow the helper, but may not contradict it. + terms = getattr(product, "term_fields", {}) or {} + for term_name, expected in ( + ("model", "black"), ("mc_distribution", process_family), ("sampling", "antithetic"), + ): + if term_name in terms and terms[term_name] != expected: + failures.append(f"unsupported_forward_marginal_profile:{term_name}") + + expected_sections = ( + ("state_spec", { + "state_variable": "forward_rate", "dimension": 1, + "state_tags": (marginal_tag,), "state_layout": "scalar", + }), + ("process_spec", { + "process_family": process_family, "simulation_scheme": "exact_lognormal", + "process_tags": ("antithetic", "no_joint_forward_process"), + }), + ("path_requirement_spec", { + "requirement_kind": marginal_tag, "snapshot_schedule_role": "observation_dates", + "reducer_kinds": ("period_option_cashflow_strip",), + "replay_mode": "independent_periods", "stored_fields": ("forward_rate",), + }), + ("payoff_reducer_spec", { + "reducer_kind": "period_option_cashflow_strip", + "output_semantics": "period_rate_option_strip_payoff", + }), + ("control_spec", {"control_style": "identity", "controller_role": "none"}), + ("measure_spec", { + "measure_family": "period_payment_forward", + "numeraire_binding": "payment_date_discount_factor", + }), + ) + for section_name, expected_fields in expected_sections: + section = getattr(family_ir, section_name, None) + if any(getattr(section, field, None) != expected for field, expected in expected_fields.items()): + failures.append(f"unsupported_forward_marginal_profile:{section_name}") + if calibration_step is not None or family_ir.calibration_binding is not None: + failures.append("unsupported_forward_marginal_profile:calibration") + return tuple(failures) + + def _has_automatic_events(product) -> bool: """Return whether the semantic product relies on automatic event transitions.""" controller_style = str(getattr(getattr(product, "controller_protocol", None), "controller_style", "identity")).strip() or "identity"