@@ -14767,30 +14767,39 @@ def load_benchmark_corpus(path: Path) -> list[dict]:
1476714767# "fewer tokens, same answer" must be able to fail a build when the answer
1476814768# stops being reachable. These are the floors CI enforces.
1476914769#
14770- # These are calibrated to MEASURED performance on the shipped corpora, with a
14771- # regression margin — they are detectors, not aspirations. Numbers observed at
14772- # the product default budget of 6000 tokens across 96 cases in 4 repositories
14773- # (Python / TypeScript / Go):
14770+ # CRITICAL — these floors must match the BACKEND the gate actually runs on.
14771+ # CI runs `benchmark --check` with TOKENGRAPH_EMBEDDINGS=off (see ci.yml): the
14772+ # deterministic hash backend, chosen so every runner scores identically with no
14773+ # model download and no network dependency. The hash backend captures lexical
14774+ # and structural overlap but does NOT match by meaning across unrelated
14775+ # vocabulary, so it scores lower than the neural backend on the same corpora —
14776+ # and several self-corpus cases are deliberately paraphrased away from the code's
14777+ # own words, which only true embeddings can bridge.
1477414778#
14775- # recall_at_5 0.979 | symbol_recall 0.865 | answerable 0.698 | waste 0.717
14779+ # Measured at the product default budget of 6000 tokens across 96 cases in 4
14780+ # repositories (Python / TypeScript / Go), on the two backends:
1477614781#
14777- # Previously 0.958 / 0.719 / 0.510 / 0.785. The gain came from two fixes, both
14778- # aimed at the same finding — that retrieval located the right file and then
14779- # failed to carry the thing in it that answered the question:
14782+ # embeddings ON (sentence-transformers):
14783+ # recall_at_5 0.979 | symbol_recall 0.865 | answerable 0.698 | waste 0.717
14784+ # embeddings OFF (hash backend — what CI runs):
14785+ # recall_at_5 0.958 | symbol_recall 0.766 | answerable 0.625 | waste 0.745
1478014786#
14781- # * SR-1, the completion sweep, which stopped packs terminating at half the
14782- # requested budget with the answer left on the floor; and
14783- # * CN-1, indexing module- and type-scope constants, without which a
14784- # controlling value was not a symbol and could not be retrieved at all.
14787+ # The earlier floors (recall 0.95 / symbol 0.80) were set from the embeddings-ON
14788+ # numbers while CI ran the hash backend, so the gate could never pass as
14789+ # configured (the hash backend scored 0.781 symbol_recall even at the commit that
14790+ # recorded 0.865). The floors below are calibrated to the hash-backend baseline
14791+ # with a regression margin — detectors, not aspirations, for the environment the
14792+ # gate actually measures. If CI is ever switched to install and warm the neural
14793+ # backend, raise these back toward the embeddings-ON row.
1478514794#
14786- # The honest reading is still that answerability is the weak metric: roughly a
14787- # third of packs remain short of some symbol or literal an answer needs. Do not
14788- # raise a threshold without first raising the measurement.
14795+ # The honest reading is that answerability is the weak metric: roughly a third of
14796+ # packs remain short of some symbol or literal an answer needs. Do not raise a
14797+ # threshold without first raising the measurement — on the SAME backend CI runs .
1478914798BENCHMARK_THRESHOLDS = {
14790- "recall_at_5": 0.95 , # the right file is in the top 5
14791- "symbol_recall": 0.80 , # required symbols actually made it into the pack
14792- "answerable_rate": 0.60, # packs carrying EVERY required symbol + fact
14793- "irrelevant_token_ratio": 0.78, # ceiling — budget spent outside target files
14799+ "recall_at_5": 0.93 , # the right file is in the top 5 (hash: 0.958)
14800+ "symbol_recall": 0.74 , # required symbols made it into the pack (hash: 0.766)
14801+ "answerable_rate": 0.60, # packs carrying EVERY required symbol + fact (hash: 0.625)
14802+ "irrelevant_token_ratio": 0.78, # ceiling — budget spent outside target files (hash: 0.745)
1479414803}
1479514804
1479614805
0 commit comments