Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
178 changes: 94 additions & 84 deletions contracts/model/decision-questions.json
Original file line number Diff line number Diff line change
Expand Up @@ -17,6 +17,11 @@
"train": 382,
"validation": 103
},
"closure.outcome": {
"test": 115,
"train": 382,
"validation": 103
},
"closure.withdrawn": {
"test": 115,
"train": 382,
Expand All @@ -26,24 +31,29 @@
"model": "typesafe/jev-1.13",
"test": {
"closure.deadline_changed": {
"accuracy": 0.9565217391304348,
"brier": 0.04520347826086957,
"ece_10": 0.08695652173913043
"accuracy": 0.9304347826086956,
"brier": 0.06380869565217391,
"ece_10": 0.12939130434782606
},
"closure.fulfilled": {
"accuracy": 0.8695652173913043,
"brier": 0.1078886956521739,
"ece_10": 0.15686956521739134
"brier": 0.09252521739130433,
"ece_10": 0.13947826086956522
},
"closure.modified": {
"accuracy": 0.4782608695652174,
"brier": 0.33723652173913055,
"ece_10": 0.35113043478260875
"accuracy": 0.5043478260869565,
"brier": 0.33256173913043485,
"ece_10": 0.31530434782608696
},
"closure.outcome": {
"accuracy": 0.8260869565217391,
"brier": 0.2776495652173913,
"ece_10": 0.10530434782608707
},
"closure.withdrawn": {
"accuracy": 1.0,
"brier": 0.01063304347826087,
"ece_10": 0.08069565217391303
"brier": 0.014983478260869564,
"ece_10": 0.09573913043478262
}
}
},
Expand Down Expand Up @@ -83,34 +93,34 @@
"model": "typesafe/jev-1.13",
"test": {
"rules.deadline_kind": {
"accuracy": 0.7349397590361446,
"brier": 0.42919518072289153,
"ece_10": 0.19192771084337357
"accuracy": 0.7590361445783133,
"brier": 0.3796602409638554,
"ece_10": 0.15795180722891577
},
"rules.duplicate_action": {
"accuracy": 0.9012345679012346,
"brier": 0.08698271604938271,
"ece_10": 0.11061728395061728
"accuracy": 0.9135802469135802,
"brier": 0.08201111111111112,
"ece_10": 0.09345679012345678
},
"rules.event_match": {
"accuracy": 0.8235294117647058,
"brier": 0.13567176470588233,
"ece_10": 0.11494117647058823
"accuracy": 0.788235294117647,
"brier": 0.16230352941176474,
"ece_10": 0.16776470588235287
},
"rules.recap": {
"accuracy": 1.0,
"brier": 0.014240740740740738,
"ece_10": 0.09320987654320988
"brier": 0.0151,
"ece_10": 0.09444444444444446
},
"rules.scoped_event": {
"accuracy": 0.8928571428571429,
"brier": 0.08740357142857143,
"ece_10": 0.11821428571428574
"brier": 0.08850833333333333,
"ece_10": 0.11845238095238095
},
"rules.thread_merge": {
"accuracy": 0.9054054054054054,
"brier": 0.08270810810810811,
"ece_10": 0.10864864864864865
"brier": 0.05593783783783782,
"ece_10": 0.08297297297297303
}
}
},
Expand Down Expand Up @@ -150,34 +160,34 @@
"model": "typesafe/jev-1.13",
"test": {
"triage.asks_question": {
"accuracy": 0.8051948051948052,
"brier": 0.1256051948051948,
"ece_10": 0.17636363636363633
"accuracy": 0.7532467532467533,
"brier": 0.15356883116883116,
"ece_10": 0.19662337662337664
},
"triage.asks_recipient": {
"accuracy": 0.9102564102564102,
"brier": 0.05079358974358975,
"ece_10": 0.10653846153846154
"accuracy": 0.9487179487179487,
"brier": 0.037028205128205136,
"ece_10": 0.09461538461538474
},
"triage.automated_notification": {
"accuracy": 0.9887640449438202,
"brier": 0.016422471910112358,
"ece_10": 0.04786516853932592
"brier": 0.014239325842696628,
"ece_10": 0.03831460674157311
},
"triage.boilerplate": {
"accuracy": 0.9864864864864865,
"brier": 0.006682432432432433,
"ece_10": 0.05337837837837836
"brier": 0.009335135135135135,
"ece_10": 0.05837837837837828
},
"triage.commits_sender": {
"accuracy": 0.8625,
"brier": 0.09970500000000002,
"ece_10": 0.15899999999999995
"accuracy": 0.85,
"brier": 0.11351625000000001,
"ece_10": 0.18037500000000004
},
"triage.names_time": {
"accuracy": 0.8472222222222222,
"brier": 0.12589166666666665,
"ece_10": 0.18000000000000002
"accuracy": 0.9027777777777778,
"brier": 0.07234722222222222,
"ece_10": 0.09555555555555552
}
}
}
Expand All @@ -202,39 +212,39 @@
"type": "choice"
},
"closure.deadline_changed": {
"accept": 0.7,
"escalate": 0.65,
"accept": 0.95,
"escalate": 0.5,
"instructions": "The later text sets a different deadline for the obligation.",
"type": "noul"
},
"closure.fulfilled": {
"accept": 0.7,
"escalate": 0.65,
"accept": 0.93,
"escalate": 0.5,
"instructions": "The later text shows the obligation has been carried out.",
"type": "noul"
},
"closure.modified": {
"accept": 0.7,
"accept": 0.98,
"escalate": 0.2,
"instructions": "The later text changes what the obligation requires.",
"type": "noul"
},
"closure.outcome": {
"accept": 0.7,
"escalate": 0.3,
"accept": 1.0,
"escalate": 0.8,
"instructions": "Which single outcome, if any, does the later paragraph express for the obligation?",
"options": {
"fulfilled": "The later paragraph says the obligation was carried out.",
"withdrawn": "The later paragraph cancels or withdraws the obligation.",
"deadline_changed": "The later paragraph sets a different deadline for the obligation.",
"fulfilled": "The later paragraph says the obligation was carried out.",
"modified": "The later paragraph changes what the obligation requires.",
"none": "The later paragraph expresses none of the listed outcomes for the obligation."
"none": "The later paragraph expresses none of the listed outcomes for the obligation.",
"withdrawn": "The later paragraph cancels or withdraws the obligation."
},
"type": "choice"
},
"closure.withdrawn": {
"accept": 0.5,
"escalate": 0.45,
"escalate": 0.33,
"instructions": "The later text cancels or withdraws the obligation.",
"type": "noul"
},
Expand All @@ -243,12 +253,12 @@
"escalate": 0.3,
"instructions": "Which single claim type, if any, does the paragraph express?",
"options": {
"request": "The paragraph asks its recipient to do something.",
"promise": "The paragraph commits its sender to doing something.",
"question": "The paragraph asks a genuine question that seeks an answer.",
"attribution": "The paragraph attributes an obligation or commitment to another party.",
"delegation": "The paragraph delegates an obligation from one party to another.",
"none": "The paragraph expresses none of the listed claim types."
"none": "The paragraph expresses none of the listed claim types.",
"promise": "The paragraph commits its sender to doing something.",
"question": "The paragraph asks a genuine question that seeks an answer.",
"request": "The paragraph asks its recipient to do something."
},
"type": "choice"
},
Expand All @@ -271,9 +281,9 @@
"type": "choice"
},
"rules.deadline_kind": {
"accept": 0.7,
"accept": 1.0,
"escalate": 0.65,
"instructions": "Classify how the phrase expresses a deadline.",
"instructions": "The phrase is classified as event-tied, soft, or unknown regarding its deadline timing.",
"options": {
"event_tied": "The deadline is tied to a named event.",
"soft": "The timing is flexible or aspirational.",
Expand All @@ -282,72 +292,72 @@
"type": "choice"
},
"rules.duplicate_action": {
"accept": 0.7,
"escalate": 0.65,
"instructions": "The action sentence `action_a` asks for the same thing as the action sentence `action_b`.",
"accept": 0.9,
"escalate": 0.36,
"instructions": "action_a and action_b are semantically equivalent.",
"type": "noul"
},
"rules.event_match": {
"accept": 0.7,
"escalate": 0.6,
"instructions": "The phrase refers to the event name.",
"accept": 0.96,
"escalate": 0.5,
"instructions": "The phrase refers to the event_name.",
"type": "noul"
},
"rules.recap": {
"accept": 0.5,
"escalate": 0.45,
"escalate": 0.4,
"instructions": "This email is an automatically generated meeting summary, recap, or transcript.",
"type": "noul"
},
"rules.scoped_event": {
"accept": 0.85,
"escalate": 0.75,
"accept": 0.91,
"escalate": 0.395,
"instructions": "This request is about attending, preparing for, or bringing something to a meeting or event.",
"type": "noul"
},
"rules.thread_merge": {
"accept": 0.85,
"escalate": 0.8,
"instructions": "The topic of first_paragraph_a is the topic of first_paragraph_b.",
"accept": 0.915,
"escalate": 0.5,
"instructions": "The conversation topic of first_paragraph_a is identical to that of first_paragraph_b.",
"type": "noul"
},
"triage.asks_question": {
"accept": 0.95,
"escalate": 0.9,
"instructions": "The paragraph_text asks a genuine question that seeks an answer.",
"escalate": 0.88,
"instructions": "The paragraph_text asks a genuine question seeking an answer.",
"type": "noul"
},
"triage.asks_recipient": {
"accept": 0.5,
"escalate": 0.45,
"instructions": "The paragraph_text explicitly instructs its recipient to perform an action.",
"accept": 0.9,
"escalate": 0.8,
"instructions": "The paragraph_text directly requests the recipient to perform an action.",
"type": "noul"
},
"triage.automated_notification": {
"accept": 0.5,
"escalate": 0.45,
"escalate": 0.4,
"instructions": "The paragraph is an automatically generated notification.",
"type": "noul"
},
"triage.boilerplate": {
"accept": 0.5,
"escalate": 0.45,
"escalate": 0.4,
"instructions": "The paragraph is a signature, legal footer, unsubscribe notice, or disclaimer.",
"type": "noul"
},
"triage.commits_sender": {
"accept": 0.9,
"escalate": 0.85,
"instructions": "The paragraph commits its sender to doing something.",
"accept": 0.93,
"escalate": 0.5,
"instructions": "The paragraph_text commits the user to an action.",
"type": "noul"
},
"triage.names_time": {
"accept": 0.7,
"escalate": 0.65,
"instructions": "The paragraph names a time, date, deadline, or event-relative time.",
"accept": 0.955,
"escalate": 0.5,
"instructions": "The paragraph_text names a time, date, or deadline.",
"type": "noul"
}
},
"schema_version": 1,
"tuned_at": "2026-09-20T21:50:51.564400+00:00"
"tuned_at": "2026-09-27T15:18:48.313117+00:00"
}
2 changes: 1 addition & 1 deletion tools/check-model-boundary.ps1
Original file line number Diff line number Diff line change
Expand Up @@ -287,7 +287,7 @@ $decisionQuestions = Read-Json $DecisionQuestionsPath 'P0-MODEL-DECISIONS-001'
if ($null -ne $decisionQuestions) {
$decisionCanonical = $decisionQuestions | ConvertTo-Json -Depth 100 -Compress
$decisionHash = Sha256-Hex $decisionCanonical
if ($decisionHash -ne '4bd00f0347538f3e91865079e36f70bee8e814685e92382840e22cea6c763ce2' -or $decisionQuestions.schema_version -ne 1 -or $decisionQuestions.model -ne 'typesafe/jev-1.13') { Fail 'P0-MODEL-DECISIONS-001' }
if ($decisionHash -ne '2a4323b958ba671f35791be4a8392e6c831b368c0cae23420eaf9a3c1c122539' -or $decisionQuestions.schema_version -ne 1 -or $decisionQuestions.model -ne 'typesafe/jev-1.13') { Fail 'P0-MODEL-DECISIONS-001' }
$questionsProperty = $decisionQuestions.PSObject.Properties['questions']
if ($null -eq $questionsProperty -or @($questionsProperty.Value.PSObject.Properties).Count -eq 0) {
Fail 'P0-MODEL-DECISIONS-001'
Expand Down
34 changes: 25 additions & 9 deletions tools/jev-optimize/README.md
Original file line number Diff line number Diff line change
Expand Up @@ -24,6 +24,12 @@ the model that authors synthetic rows. The Decisions endpoint receives
`provider: {"zdr": true}` unless a content-free probe shows that the alpha
endpoint rejects that member. The reflection endpoint always requests ZDR.

`JevLM` sends the production wire shape: flat state fields, the registry question
ID as the question key, and the DSPy signature docstring in that question's
`instructions`. `tests/test_offline_end_to_end.py` runs the real GEPA and
ReAnchor path against a fake Decisions endpoint and is the required offline gate
before any paid optimization run.

## Commands

```powershell
Expand Down Expand Up @@ -87,9 +93,12 @@ possible among `event_tied`, `soft`, and `unknown` (34/33/33 at 100 rows).
For every applicable binary label, `from_user` differs by at most 0.1 between
positive and negative rows and the `days_later` means differ by at most 2 days.
Each corpus has at least 300 distinct paragraph texts, and closure has at least
40 distinct obligation texts. The threshold tuner requires at least 20 positives
and 20 negatives in its sweep;
smaller samples retain the registry defaults.
40 distinct obligation texts. Calibration requires at least 20 positives and
20 negatives; smaller samples retain the registry defaults. For Noul questions,
DSPy's ReAnchor runs twice on train plus validation after GEPA: one metric heavily
penalizes false positives to fit the accept threshold, and the mirrored metric
heavily penalizes false negatives to fit the escalation threshold. Choice
questions continue to use the confidence-based selective-classification sweep.

LLM generation uses a fixed matrix of at least 30 scenario seeds per question,
sends a separate prompt and strict schema for each question batch, assigns
Expand Down Expand Up @@ -138,9 +147,16 @@ already outside this repository by construction (the exporter refuses to
write inside it), so no extra step is needed to keep them out of `git`.

The registry hash printed by `write-registry` is SHA-256 over
`json.dumps(obj, separators=(",", ":"), ensure_ascii=False)`. The registry is
written with sorted keys, so PowerShell's `ConvertTo-Json -Depth 100 -Compress`
preserves the same member order. On the hand-written bootstrap file the hashes
do differ: PowerShell preserves the decimal scale (`0.70`) while Python emits
`0.7`. After `write-registry` normalizes the file through Python, the two
recipes serialize parsed numeric values identically.
`json.dumps(obj, separators=(",", ":"), ensure_ascii=False)`. It is not the
value to pin in `tools/check-model-boundary.ps1`. That checker hashes
`ConvertTo-Json -Depth 100 -Compress` of the parsed file, and PowerShell parses
`tuned_at` into a DateTime and re-emits it in the machine's local offset, so the
two recipes differ in that one member (and the checker's value depends on the
timezone of the machine that computes it). Pin the checker's own value:

```powershell
$c = Get-Content -Raw contracts/model/decision-questions.json | ConvertFrom-Json
$json = $c | ConvertTo-Json -Depth 100 -Compress
$sha = [System.Security.Cryptography.SHA256]::Create()
($sha.ComputeHash([Text.Encoding]::UTF8.GetBytes($json)) | ForEach-Object { $_.ToString('x2') }) -join ''
```
Loading
Loading