From dd8917cf892ae33dac35cdd2c879c6e0e0bd02a8 Mon Sep 17 00:00:00 2001 From: Daniel W Liu <117604131+LTKers@users.noreply.github.com> Date: Thu, 3 Sep 2026 22:46:45 -0400 Subject: [PATCH 1/2] fix(cli): --speed banana silently replayed unpaced, which is the wrong measurement Found by taking a defect out of the sibling repo and asking whether this one has it. Voxel's argument parser used std::atoi on five count flags, so a typo parsed to zero and silently disabled the feature. This repo's parser is better - parse_int uses from_chars and requires the whole string consumed, flag_value returns nullopt on malformed input, and callers check it - so the answer was mostly no. flag_double was the exception. It returned the FALLBACK on unparseable input and on a flag given with no value at all, so: basis replay capture.feedlog --speed banana basis replay capture.feedlog --speed both fell back to 0.0, which means "replay flat out". No error, no warning, and a normal-looking report. That is worse here than the equivalent was in voxel. --speed 1 exists to replay at the venue's real arrival schedule and measure response time instead of service time - docs/bench/latency.md's whole finding is that the service p99 understates what a consumer waits by 13.3x. A typo therefore returned exactly the service-time numbers that mode was built to correct, in the benchmark built to correct them, and the only tell was the absence of one line in the output. flag_double now returns std::optional like flag_value, signalling rather than substituting, and both call sites reject with a message naming the flag and the expected shape. The missing-value case is handled explicitly rather than falling out of the loop bound, since that is how it went unnoticed. --speed banana -> exit 1 --speed 1x -> exit 1 (trailing junk) --speed -> exit 1 (no value) --speed -1 -> exit 1 (negative) --pace-spin-ms xyz -> exit 1 --speed 0 and --speed 1 unchanged; 228 tests, the perf gate and the bench-input check all pass. --- src/cli/args.cpp | 14 +++++++++----- src/cli/args.h | 7 ++++++- src/cli/replay_commands.cpp | 14 ++++++++++++-- 3 files changed, 27 insertions(+), 8 deletions(-) diff --git a/src/cli/args.cpp b/src/cli/args.cpp index fece641..63dc791 100644 --- a/src/cli/args.cpp +++ b/src/cli/args.cpp @@ -33,18 +33,22 @@ std::string flag_string(const std::vector& args, return fallback; } -double flag_double(const std::vector& args, - std::string_view name, double fallback) { - for (std::size_t i = 0; i + 1 < args.size(); ++i) { +std::optional flag_double(const std::vector& args, + std::string_view name, double fallback) { + for (std::size_t i = 0; i < args.size(); ++i) { if (args[i] != name) continue; + // Present as the final argument, so there is no value to read. Silently + // falling back here is what let `--speed` with no number look like a + // deliberate unpaced run. + if (i + 1 >= args.size()) return std::nullopt; const std::string text(args[i + 1]); try { std::size_t consumed = 0; const double v = std::stod(text, &consumed); - if (consumed != text.size()) return fallback; // trailing junk + if (consumed != text.size()) return std::nullopt; // trailing junk return v; } catch (...) { - return fallback; + return std::nullopt; } } return fallback; diff --git a/src/cli/args.h b/src/cli/args.h index 20393c4..089cd45 100644 --- a/src/cli/args.h +++ b/src/cli/args.h @@ -35,7 +35,12 @@ std::string flag_string(const std::vector& args, // a rate, and every rate they accept is validated against its own bounds // right after, so a malformed value is caught there with a message that // names the flag. -double flag_double(const std::vector& args, +// Returns nullopt when the flag is present but its value is not a +// complete number, or when it is present with no value at all - the same +// contract flag_value has, and for the same reason. It used to return the +// fallback in both cases, so `--speed banana` silently produced an +// unpaced replay: the one measurement that mode exists to correct. +std::optional flag_double(const std::vector& args, std::string_view name, double fallback); // Comma-separated flag values (--binance btcusdt,ethusdt). diff --git a/src/cli/replay_commands.cpp b/src/cli/replay_commands.cpp index 28bfb8a..a57540c 100644 --- a/src/cli/replay_commands.cpp +++ b/src/cli/replay_commands.cpp @@ -121,11 +121,21 @@ int run_replay(const std::vector& args) { // factor. 0 keeps the historical flat-out replay, which can only ever // report service time. const auto speed = flag_double(args, "--speed", 0.0); + if (!speed || *speed < 0.0 || *speed > 100000.0) { + basis::log::error("replay: --speed expects a non-negative number " + "(0 replays flat out, 1 replays at the venue's rate)"); + return usage(); + } // How much of each pacing wait is spun instead of slept. Past the // capture's largest gap this spins every wait, which is what it takes // for the harness's own timer error to stop dominating a // microsecond-scale response time. const auto spin_ms = flag_double(args, "--pace-spin-ms", 2.0); + if (!spin_ms || *spin_ms < 0.0 || *spin_ms > 60000.0) { + basis::log::error("replay: --pace-spin-ms expects a non-negative number " + "of milliseconds"); + return usage(); + } const auto episodes_csv_path = flag_string(args, "--episodes-csv", ""); std::string error; @@ -196,8 +206,8 @@ int run_replay(const std::vector& args) { basis::bench::ReplayHarness harness(*registry, &session, {}, book_mr); harness.set_parse_arena(parse_arena); harness.set_breakdown(breakdown); - harness.set_replay_speed(speed); - harness.set_pace_spin_ns(static_cast(spin_ms * 1e6)); + harness.set_replay_speed(*speed); + harness.set_pace_spin_ns(static_cast(*spin_ms * 1e6)); const auto stats = harness.run(in_path, &error); if (!stats) { basis::log::error(error); From 0b1c488dfa8f7d42b3708029c35aeb078ad395b1 Mon Sep 17 00:00:00 2001 From: Daniel W Liu <117604131+LTKers@users.noreply.github.com> Date: Thu, 3 Sep 2026 22:47:29 -0400 Subject: [PATCH 2/2] docs: audit what is left, and rank the three gaps that are real Measured first. Test depth is this repo's strength and it is not close: 5,357 test lines over 10,828 source, a ratio of 0.49 against the sibling repo's 0.21, with every core library covered. The components with no tests are cli/ command wiring, which is mostly orchestration, and the one place that gap had teeth has just been closed. Ranked the three gaps that are real. Kalshi has never run live and the blocker is credentials rather than code, which makes it the only item that cannot be done by writing any, and the one that unlocks the most - with it the arbitrage backtester runs on a real both-venue capture instead of a synthetic session. Trades are absent and the engine carries only price levels, which is bigger than it looks because the live feed subscribes for depth, so it needs a channel change and a fresh capture rather than a parser change. And there is no storage layer at all. Also records what is explicitly not on the list and why, so the same proposals do not have to be re-litigated: Kafka (a milliseconds tool against a microsecond headline, and 1,170 repos), a fourth lead-lag estimator (three already agree), and chasing throughput (already four orders of magnitude above the venue). --- docs/roadmap.md | 84 +++++++++++++++++++++++++++++++++++++++++++++++++ 1 file changed, 84 insertions(+) create mode 100644 docs/roadmap.md diff --git a/docs/roadmap.md b/docs/roadmap.md new file mode 100644 index 0000000..d4d3a39 --- /dev/null +++ b/docs/roadmap.md @@ -0,0 +1,84 @@ +# What is left here + +An audit on 2026-09-03, run the same way as the sibling repo's: measure +first, then decide. + +## Where this repo is strong + +Test depth, and it is not close: + + basis 5,357 test lines over 10,828 source ratio 0.49 + voxel 1,993 test lines over 9,313 source ratio 0.21 + +Every core library - feed, model, analytics, api, exec, normalize, core - +has tests. The components with none are `cli/` command wiring and +`logger.cpp`, and CLI commands are mostly orchestration: open a file, call +a library, print. There is real value in testing them, and it is a long +way below the value of the gaps named next. + +Argument parsing was the one place the CLI gap had teeth, and it has been +closed: `flag_double` used to substitute a fallback for unparseable input, +so `--speed banana` silently replayed unpaced. + +## The three real gaps, ranked + +### 1. Kalshi has never run live + +The blocker is credentials, not code. The adapter exists and is verified +offline down to the RSA-PSS signature. `configs/contracts.toml` maps 14 +cross-venue contracts. None of it has ever been exercised against the +venue, so every cross-venue result in this repo is Binance/Coinbase - the +substitute pairing, chosen because it is public on both sides. + +The cost is a free account and an RSA key. It is the only item on this +list that cannot be done by writing code, and it is the one that unlocks +the most: with it, the fee-aware arbitrage backtester runs on a real +both-venue capture instead of a synthetic session, and the repo's stated +thesis becomes a measured result rather than a described one. + +### 2. This is a book engine, not a market-data engine + +`BookDelta` carries price levels. `Action` is Set, Add, Clear. There are +no trade prints, no last-trade, no volume - and every venue here publishes +them on the same socket. + +The Coinbase parser can already read a `ticker` frame, which carries +trades. The live feed subscribes `level2_batch` instead, for depth, so no +committed capture contains a single trade. That is what makes this bigger +than it looks: it needs a channel change and a fresh capture, not a parser +change. + +Worth it because it is half of what a market-data feed carries, and +because it would give `ConflatingSession` its counter-example. That doc +already argues fills and prints want a queue rather than conflation, and +there is nothing in the repo to point at. + +### 3. There is no storage layer + +`.feedlog` is one record per line, tab-separated, raw JSON, gzipped. That +is a capture format. There is no index, no columnar layout, no time-range +query; reading a thirty-minute window means scanning the file. + +This is the largest piece of work on the list and the one furthest from +what exists. It is also the only one that would add a capability the repo +does not gesture at anywhere else. + +## What is explicitly not on the list + +**Kafka, for replay or event streaming.** Considered and rejected on +numbers. Kafka is a milliseconds tool and this engine's headline is a +microsecond service time; putting it in the path deletes the number the +repo is built on, and putting it outside the path leaves it doing nothing. +`kafka market data` is also 1,170 GitHub repositories, which is the +default tutorial architecture rather than a differentiator. The gap it +would fill - cross-process fan-out - is real, and the domain-correct +answer to it is a shared-memory ring buffer, which is what CME and +Chronicle actually use and which preserves the latency story instead of +destroying it. + +**More estimators.** Three already agree on the cross-venue ordering: +cross-correlation, an event study, and Hayashi-Yoshida. A fourth would not +make the finding more true. + +**Chasing throughput.** ~2M messages/sec against a venue producing 269. +The ratio is the point and it is already four orders of magnitude.