diff --git a/.gitignore b/.gitignore index ea8c4bf..73723a6 100644 --- a/.gitignore +++ b/.gitignore @@ -1 +1,2 @@ /target +/runs diff --git a/Cargo.lock b/Cargo.lock index 5c4ef72..d8b6c4a 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -923,12 +923,34 @@ dependencies = [ "tracing-subscriber", ] +[[package]] +name = "base-x" +version = "0.2.11" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "4cbbc9d0964165b47557570cce6c952866c2678457aca742aafc9fb771d30270" + [[package]] name = "base16ct" version = "0.2.0" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "4c7f02d4ea65f2c1853089ffd8d2787bdbc63de2f0d29dedbcf8ccdfa0ccd4cf" +[[package]] +name = "base256emoji" +version = "1.0.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "b5e9430d9a245a77c92176e649af6e275f20839a48389859d1661e9a128d077c" +dependencies = [ + "const-str", + "match-lookup", +] + +[[package]] +name = "base45" +version = "3.2.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "240e56f4d3c453c36faacb695c535a4d5f8c7d23dac175014f32eb0a71012a03" + [[package]] name = "base64" version = "0.21.7" @@ -1154,6 +1176,15 @@ version = "0.3.0" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "37b2a672a2cb129a2e41c10b1224bb368f9f37a2b16b612598138befd7b37eb5" +[[package]] +name = "cbor4ii" +version = "0.2.14" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "b544cf8c89359205f4f990d0e6f3828db42df85b5dac95d09157a250eb0749c4" +dependencies = [ + "serde", +] + [[package]] name = "cc" version = "1.2.56" @@ -1241,6 +1272,19 @@ dependencies = [ "half", ] +[[package]] +name = "cid" +version = "0.11.3" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "21a304f95f84d169a6f31c4d0a30d784643aaa0bbc9c1e449a2c23e963ec4971" +dependencies = [ + "multibase", + "multihash", + "serde", + "serde_bytes", + "unsigned-varint", +] + [[package]] name = "cipher" version = "0.4.4" @@ -1614,6 +1658,12 @@ version = "0.9.6" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "c2459377285ad874054d797f3ccebf984978aa39129f6eafde5cdc8315b612f8" +[[package]] +name = "const-str" +version = "0.4.3" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "2f421161cb492475f1661ddc9815a745a1c894592070661180fdec3d4872e9c3" + [[package]] name = "const_format" version = "0.2.35" @@ -1835,6 +1885,33 @@ dependencies = [ "thiserror 1.0.69", ] +[[package]] +name = "crypto" +version = "0.5.0" +source = "git+https://github.com/sourcenetwork/defradb.rs?rev=8d8bb299f#8d8bb299f0453e4fedb743854f2e853605f31f61" +dependencies = [ + "aes-gcm", + "blst", + "cid", + "defra-core", + "ed25519-dalek", + "hex", + "hkdf", + "hmac", + "k256", + "multibase", + "p256", + "rand 0.8.5", + "serde", + "serde_bytes", + "sha2 0.10.9", + "subtle", + "thiserror 2.0.18", + "unsigned-varint", + "x25519-dalek", + "zeroize", +] + [[package]] name = "crypto-bigint" version = "0.5.5" @@ -1972,6 +2049,51 @@ version = "2.10.0" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "d7a1e2f27636f116493b8b860f5546edb47c8d8f8ea73e1d2a20be88e28d1fea" +[[package]] +name = "data-encoding-macro" +version = "0.1.19" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "8142a83c17aa9461d637e649271eae18bf2edd00e91f2e105df36c3c16355bdb" +dependencies = [ + "data-encoding", + "data-encoding-macro-internal", +] + +[[package]] +name = "data-encoding-macro-internal" +version = "0.1.17" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "7ab67060fc6b8ef687992d439ca0fa36e7ed17e9a0b16b25b601e8757df720de" +dependencies = [ + "data-encoding", + "syn 1.0.109", +] + +[[package]] +name = "defra-core" +version = "0.5.0" +source = "git+https://github.com/sourcenetwork/defradb.rs?rev=8d8bb299f#8d8bb299f0453e4fedb743854f2e853605f31f61" +dependencies = [ + "async-trait", + "bytes", + "ciborium", + "cid", + "ipld-core", + "multibase", + "multihash", + "rand 0.8.5", + "serde", + "serde_bytes", + "serde_ipld_dagcbor", + "serde_json", + "sha2 0.10.9", + "thiserror 2.0.18", + "tokio", + "tracing", + "uuid", + "zeroize", +] + [[package]] name = "defra-harness" version = "0.1.0" @@ -1998,6 +2120,7 @@ source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "e7c1832837b905bbfb5101e07cc24c8deddf52f93225eee6ead5f4d63d53ddcb" dependencies = [ "const-oid", + "pem-rfc7468", "zeroize", ] @@ -2117,6 +2240,7 @@ source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "115531babc129696a58c64a4fef0a8bf9e9698629fb97e9e40767d235cfbcd53" dependencies = [ "pkcs8", + "serde", "signature", ] @@ -2181,6 +2305,7 @@ dependencies = [ "ff", "generic-array", "group", + "pem-rfc7468", "pkcs8", "rand_core 0.6.4", "sec1", @@ -3077,6 +3202,23 @@ version = "1.0.1" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "b9e0384b61958566e926dc50660321d12159025e767c18e043daf26b70104c39" +[[package]] +name = "identity" +version = "0.5.0" +source = "git+https://github.com/sourcenetwork/defradb.rs?rev=8d8bb299f#8d8bb299f0453e4fedb743854f2e853605f31f61" +dependencies = [ + "base64 0.22.1", + "crypto 0.5.0", + "defra-core", + "hex", + "k256", + "p256", + "serde", + "serde_json", + "thiserror 2.0.18", + "web-time", +] + [[package]] name = "idna" version = "1.1.0" @@ -3166,6 +3308,17 @@ dependencies = [ "generic-array", ] +[[package]] +name = "ipld-core" +version = "0.4.3" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "090f624976d72f0b0bb71b86d58dc16c15e069193067cb3a3a09d655246cbbda" +dependencies = [ + "cid", + "serde", + "serde_bytes", +] + [[package]] name = "ipnet" version = "2.11.0" @@ -3345,6 +3498,17 @@ dependencies = [ "syn 2.0.117", ] +[[package]] +name = "match-lookup" +version = "0.1.3" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "549e39695cc0b640f3cb378053832db3d2133422d49e8dcae5c866a2aaf1f730" +dependencies = [ + "proc-macro2", + "quote", + "syn 2.0.117", +] + [[package]] name = "matchers" version = "0.2.0" @@ -3389,6 +3553,29 @@ version = "1.12.0" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "c9be0862c1b3f26a88803c4a49de6889c10e608b3ee9344e6ef5b45fb37ad3d1" +[[package]] +name = "multibase" +version = "0.9.3" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "7e0e4a371cbf1dfd666b658ba137763edb23c45beb43cfe369b5593cd6b437b6" +dependencies = [ + "base-x", + "base256emoji", + "base45", + "data-encoding", + "data-encoding-macro", +] + +[[package]] +name = "multihash" +version = "0.19.5" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "577c63b00ad74d57e8c9aa870b5fccebf2fd64a308a5aee9f1bb88e4aea19447" +dependencies = [ + "serde", + "unsigned-varint", +] + [[package]] name = "native-tls" version = "0.2.18" @@ -3678,7 +3865,7 @@ version = "0.1.0" dependencies = [ "base64 0.22.1", "bs58", - "crypto", + "crypto 0.0.1", "defra-harness", "ed25519-dalek", "eyre", @@ -3799,6 +3986,15 @@ version = "0.8.5" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "132dca9b868d927b35b5dd728167b2dee150eb1ad686008fc71ccb298b776fca" +[[package]] +name = "pem-rfc7468" +version = "0.7.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "88b39c9bfcfc231068454382784bb460aae594343fb030d46e9f50a645418412" +dependencies = [ + "base64ct", +] + [[package]] name = "percent-encoding" version = "2.3.2" @@ -4838,6 +5034,18 @@ dependencies = [ "syn 2.0.117", ] +[[package]] +name = "serde_ipld_dagcbor" +version = "0.6.4" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "46182f4f08349a02b45c998ba3215d3f9de826246ba02bb9dddfe9a2a2100778" +dependencies = [ + "cbor4ii", + "ipld-core", + "scopeguard", + "serde", +] + [[package]] name = "serde_json" version = "1.0.149" @@ -4977,6 +5185,16 @@ dependencies = [ "cfg-if", "cpufeatures", "digest 0.10.7", + "sha2-asm", +] + +[[package]] +name = "sha2-asm" +version = "0.6.4" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "b845214d6175804686b2bd482bcffe96651bb2d1200742b712003504a2dac1ab" +dependencies = [ + "cc", ] [[package]] @@ -5049,6 +5267,23 @@ dependencies = [ "serde", ] +[[package]] +name = "soak" +version = "0.1.0" +dependencies = [ + "crypto 0.5.0", + "defra-harness", + "eyre", + "futures", + "hex", + "identity", + "rand 0.8.5", + "reqwest 0.12.28", + "serde", + "serde_json", + "tokio", +] + [[package]] name = "socket2" version = "0.5.10" @@ -5950,6 +6185,12 @@ dependencies = [ "subtle", ] +[[package]] +name = "unsigned-varint" +version = "0.8.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "eb066959b24b5196ae73cb057f45598450d2c5f71460e98c49b738086eff9c06" + [[package]] name = "untrusted" version = "0.7.1" @@ -5988,9 +6229,14 @@ checksum = "b6c140620e7ffbb22c2dee59cafe6084a59b5ffc27a8859a5f0d494b5d52b6be" [[package]] name = "uuid" -version = "1.21.0" +version = "1.26.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "b672338555252d43fd2240c714dc444b8c6fb0a5c5335e65a07bba7742735ddb" +checksum = "b5772d71c9be8a8a6ac2117d949c5b224c1b72241bb611d9a3012edcf8af7812" +dependencies = [ + "getrandom 0.4.1", + "js-sys", + "wasm-bindgen", +] [[package]] name = "valuable" diff --git a/Cargo.toml b/Cargo.toml index 3cc4e18..eae317d 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -15,6 +15,7 @@ members = [ "crates/hub-harness", "crates/orbis-harness", "crates/acp-light-client", + "crates/soak", ] [workspace.package] diff --git a/crates/defra-harness/src/cluster/builder.rs b/crates/defra-harness/src/cluster/builder.rs index 108dd3c..e1769a6 100644 --- a/crates/defra-harness/src/cluster/builder.rs +++ b/crates/defra-harness/src/cluster/builder.rs @@ -64,6 +64,7 @@ pub struct TestClusterBuilder { p2p_transport: Option, keyring: KeyringBackend, shared_se_key: Option<[u8; 32]>, + file_keyring: bool, acp_cache_ttl: Option, acp_circuit_breaker_threshold: Option, acp_circuit_breaker_reset_timeout: Option, @@ -71,6 +72,7 @@ pub struct TestClusterBuilder { acp_receipt_timeout: Option, signing_multiplier_opt_out: bool, extra_rust_args: Vec, + extra_go_args: Vec, } impl Default for TestClusterBuilder { @@ -102,6 +104,7 @@ impl TestClusterBuilder { p2p_transport: None, keyring: KeyringBackend::None, shared_se_key: None, + file_keyring: false, acp_cache_ttl: None, acp_circuit_breaker_threshold: None, acp_circuit_breaker_reset_timeout: None, @@ -109,6 +112,7 @@ impl TestClusterBuilder { acp_receipt_timeout: None, signing_multiplier_opt_out: false, extra_rust_args: Vec::new(), + extra_go_args: Vec::new(), } } @@ -133,6 +137,17 @@ impl TestClusterBuilder { self } + /// Extra flags appended to every Go node's `start` command, after the + /// managed flags (same contract as `with_extra_rust_args`). + pub fn with_extra_go_args(mut self, args: I) -> Self + where + I: IntoIterator, + S: Into, + { + self.extra_go_args.extend(args.into_iter().map(Into::into)); + self + } + pub fn rust_nodes(mut self, n: usize) -> Self { self.rust_nodes = n; self @@ -291,6 +306,15 @@ impl TestClusterBuilder { self } + /// A per-node file keyring (`--keyring-backend file --keyring-path + /// /keys`) on both runtimes, so peer identities survive + /// restarts. With the `Env` keyring a Go node presents a new peer ID on + /// every start, and any replicator pointed at it never reconnects. + pub fn with_file_keyring(mut self) -> Self { + self.file_keyring = true; + self + } + /// Seed the same 32-byte searchable-encryption key into every node's /// keyring (Go and Rust) before start, mirroring how operators provision /// the cluster-shared SE secret per node. Forces a `File` keyring backend @@ -444,7 +468,7 @@ impl TestClusterBuilder { // A cluster-shared SE key needs a File keyring both runtimes can // share; override `--no-keyring`/Env with a per-node File backend. - let keyring = if self.shared_se_key.is_some() { + let keyring = if self.shared_se_key.is_some() || self.file_keyring { KeyringBackend::File { path: rootdir.join("keys"), secret: "integration-test-secret".to_string(), @@ -544,7 +568,7 @@ impl TestClusterBuilder { // A cluster-shared SE key needs a File keyring; otherwise Go runs // with its usual `--no-keyring`. - let keyring = if self.shared_se_key.is_some() { + let keyring = if self.shared_se_key.is_some() || self.file_keyring { KeyringBackend::File { path: rootdir.join("keys"), secret: "integration-test-secret".to_string(), @@ -587,7 +611,7 @@ impl TestClusterBuilder { acp_circuit_breaker_reset_timeout: self.acp_circuit_breaker_reset_timeout, acp_request_timeout: self.acp_request_timeout, acp_receipt_timeout: self.acp_receipt_timeout, - extra_args: Vec::new(), + extra_args: self.extra_go_args.clone(), }; let mut attempt = 1; diff --git a/crates/defra-harness/src/cluster/mod.rs b/crates/defra-harness/src/cluster/mod.rs index ad8f214..deb2449 100644 --- a/crates/defra-harness/src/cluster/mod.rs +++ b/crates/defra-harness/src/cluster/mod.rs @@ -3,4 +3,4 @@ pub mod health; pub mod runtime; pub use builder::TestClusterBuilder; -pub use runtime::TestCluster; +pub use runtime::{StoppedNode, TestCluster}; diff --git a/crates/defra-harness/src/cluster/runtime.rs b/crates/defra-harness/src/cluster/runtime.rs index 720ece9..6d709a8 100644 --- a/crates/defra-harness/src/cluster/runtime.rs +++ b/crates/defra-harness/src/cluster/runtime.rs @@ -96,6 +96,20 @@ async fn reserve_until_free( /// /// Field order matters: `nodes` and `source_hub` are dropped before `run_dir`, /// ensuring processes are killed before their data directories are removed. +/// A node stopped by [`TestCluster::stop_node`]: everything needed to start +/// it again on the same ports, which stay reserved meanwhile. +pub struct StoppedNode { + index: usize, + config: NodeConfig, + kind: NodeKind, + name: String, + api_url: String, + binary_path: PathBuf, + tcp_ports: Vec, + udp_ports: Vec, + reserved: ReservedPorts, +} + pub struct TestCluster { pub nodes: Vec, source_hub: Option, @@ -183,17 +197,10 @@ impl TestCluster { .await } - /// Restart the node at `index`, reusing its rootdir and ports. - /// - /// Drops the old process (sending SIGTERM), waits briefly, then respawns - /// the same binary with the same config on the same data directory. - /// - /// The node's ports are re-reserved the moment the old process releases - /// them and held until the replacement is spawned, so nothing can be handed - /// the address while the node is down; losing the remaining boot-time race - /// is retried on the same ports. Fresh ports are not an option โ€” the - /// caller's clients and the node's peers both hold this address. - pub async fn restart_node(&mut self, index: usize, timeout: Duration) -> Result<()> { + /// Stop a node with the SIGTERM path and hold its ports so nothing can be + /// handed the address while it is down. Start it again on the same ports + /// with [`start_stopped_node`](Self::start_stopped_node). + pub async fn stop_node(&mut self, index: usize) -> Result { let old = &self.nodes[index]; let config = old.config.clone(); let kind = old.kind; @@ -220,12 +227,40 @@ impl TestCluster { ); drop(old_node); - // Take the ports back before anything else can be handed them, then - // let the node settle while they are still guarded. - let mut reserved = + // Take the ports back before anything else can be handed them. + let reserved = reserve_until_free(&name, &tcp_ports, &udp_ports, PORT_RECLAIM_TIMEOUT).await?; - tokio::time::sleep(RESTART_SETTLE).await; + Ok(StoppedNode { + index, + config, + kind, + name, + api_url, + binary_path, + tcp_ports, + udp_ports, + reserved, + }) + } + /// Start a node stopped by [`stop_node`](Self::stop_node) on its own + /// ports; losing the boot-time bind race is retried on the same ports. + pub async fn start_stopped_node( + &mut self, + stopped: StoppedNode, + timeout: Duration, + ) -> Result<()> { + let StoppedNode { + index, + config, + kind, + name, + api_url, + binary_path, + tcp_ports, + udp_ports, + mut reserved, + } = stopped; let node: Box = match kind { // Respawn from the node's configured binary (e.g. a release artifact // or a downloaded version), not the default debug workspace path โ€” @@ -272,6 +307,20 @@ impl TestCluster { Ok(()) } + + /// Restart a node on the same ports: stop, settle while the ports are + /// still guarded, start. + /// + /// The node's ports are re-reserved the moment the old process releases + /// them and held until the replacement is spawned, so nothing can be handed + /// the address while the node is down; losing the remaining boot-time race + /// is retried on the same ports. Fresh ports are not an option โ€” the + /// caller's clients and the node's peers both hold this address. + pub async fn restart_node(&mut self, index: usize, timeout: Duration) -> Result<()> { + let stopped = self.stop_node(index).await?; + tokio::time::sleep(RESTART_SETTLE).await; + self.start_stopped_node(stopped, timeout).await + } } #[cfg(test)] diff --git a/crates/defra-harness/src/lib.rs b/crates/defra-harness/src/lib.rs index 02b7f8f..3ddb6d8 100644 --- a/crates/defra-harness/src/lib.rs +++ b/crates/defra-harness/src/lib.rs @@ -14,7 +14,7 @@ pub mod sse; pub mod wasm_lens; pub use client::DefraClient; -pub use cluster::{TestCluster, TestClusterBuilder}; +pub use cluster::{StoppedNode, TestCluster, TestClusterBuilder}; pub use divergences::NodeKind; pub use fixtures::{ documents_schema_with_policy, interaction_schema_with_policy, multi_resource_policy, diff --git a/crates/soak/Cargo.toml b/crates/soak/Cargo.toml new file mode 100644 index 0000000..af43bbf --- /dev/null +++ b/crates/soak/Cargo.toml @@ -0,0 +1,23 @@ +[package] +name = "soak" +version = "0.1.0" +description = "Cross-runtime DefraDB soak driver" +edition.workspace = true +license.workspace = true +publish = false + +[dependencies] +defra-harness = { path = "../defra-harness" } +tokio.workspace = true +reqwest.workspace = true +serde.workspace = true +serde_json.workspace = true +rand.workspace = true +futures.workspace = true +eyre.workspace = true +identity = { git = "https://github.com/sourcenetwork/defradb.rs", rev = "8d8bb299f" } +crypto = { git = "https://github.com/sourcenetwork/defradb.rs", rev = "8d8bb299f" } +hex.workspace = true + +[dev-dependencies] +tokio = { workspace = true, features = ["test-util"] } diff --git a/crates/soak/README.md b/crates/soak/README.md new file mode 100644 index 0000000..6c3c3d5 --- /dev/null +++ b/crates/soak/README.md @@ -0,0 +1,367 @@ +# soak: cross-runtime DefraDB soak driver + +One binary that boots a mixed Go/Rust DefraDB mesh through `defra-harness`, +drives a seeded workload against it, keeps checking that the runtimes +converge, injects restarts and crashes on a seeded schedule, meters disk and +memory, and writes one replayable artifact directory per run. + +Status: M1b (two Rust nodes on regolith + two Go nodes on badger, on this +host, as processes, in a full replicator mesh; `p1-encrypted` profile: +encrypted fields + searchable-encryption index, checker M5). The design and +roadmap live in the agent-ops vault under `Worklogs/cross-defra/soak-harness/`. + +## Prerequisites + +- A built Rust `defra` (release recommended: `cargo build --release -p cli` + in defradb.rs), passed as `DEFRA_RUST_BINARY`. +- The Go `defradb` built at `GO_COMPAT_COMMIT` (see + `crates/defra-version/src/lib.rs` in defradb.rs) on `PATH`, with + `DEFRA_GO_COMPAT_COMMIT` set to that commit. +- `identity new` is run on both binaries at setup for `--profile p1-encrypted`, + and on the Rust binary for the owner/reader of `--profile p2-acp`. + +```sh +export DEFRA_RUST_BINARY=~/Repos/Source/defradb.rs/target/release/defra +export PATH=~/.cache/defra-harness/v1.1.0:$PATH DEFRA_GO_COMPAT_COMMIT=v1.1.0 +cargo run -p soak -- run --seed 42 --ops 1800 --rate 3 --churn +``` + +### ACP profile + +`--profile p2-acp` builds the cluster with local document ACP. Setup +generates two identities (`owner`, `reader`) with the Rust binary and +records them (key hex and DID) under `identities` in the manifest, so a +p2-acp `manifest.json` holds the two private keys in cleartext (throwaway +per-run identities, but do not paste a p2-acp manifest into an issue or +chat); the owner adds `USER_ACP_POLICY` on every node (the policy ids must agree +across runtimes or setup fails) and then the `User` schema bound to it, +and a bearer-token probe against rust-0 and go-0 must pass before the +workload starts. Creates of protected docs and all queries go over HTTP +with a bearer token for the op's actor; `grant` ops run the origin node's +own CLI (`--url host:port client -i acp document relationship add +... -r reader`) because the HTTP API has no relationship endpoint. Query +ops read the doc as owner, reader and anonymous and record the result as +`views owner= reader= anon=` in `ops.jsonl`. The checker sweeps as the +owner, and M6 compares the three views of every protected doc across each +node pair; a pair where exactly one side is the doc's origin node is +skipped and counted as `m6_by_design` (local ACP gates only there). + +Both nodes get a file keyring so their peer identities survive restarts. +The nodes' data and logs are kept under the run directory (the driver points +`DEFRA_WORKSPACE_ROOT` there and sets `DEFRA_E2E_KEEP=1`). + +## Commands + +| Command | What it does | +|---|---| +| `soak run [flags]` | A new run under `runs/-/`. | +| `soak replay --manifest /manifest.json [--until-op N] [--hold]` | Rebuilds a run from its manifest: same seed, profile, executed op count and churn schedule, no disk budget. `--until-op` stops the workload early; `--hold` keeps the mesh up until Enter, printing each node's GraphQL URL. | +| `soak summarize ` | Rewrites `profile.json` / `profile.md` from the artifact and prints the markdown. | +| `soak compare ` | Checks two runs against the replay contract; exits non-zero if they differ. | +| `soak manage --topology rg --out [--cases R2,A2,S1]` | Pass/fail cases on the P2P management channel, see "Management channel". | + +`run` flags (all optional): + +| Flag | Default | Meaning | +|---|---|---| +| `--seed N` | unix time | Master seed; both axes derive from it. | +| `--profile NAME` | p0-crud | Workload profile: `p0-crud` (plaintext Users), `p0-size` (p0-crud with a within-run payload-size mix: 256/1,200/16,000/128,000 bytes weighted 40/30/20/10; mixed-size correctness, not a disk decomposition), `p0-size-256` / `p0-size-128k` (p0-crud at a fixed create size, for one-term disk comparison against a same-day `p0-crud` run), `p0-index` (p0-crud with `@index` on `age`; the planned GraphQL is byte-identical to p0-crud at the same seed), `p3-relation` (one-to-many Author/Book; child creates use `author: ""` resolved from a parent slot), `p1-encrypted` (Vault with encrypted secret/pin and an SE index on name; builds the cluster with encryption, dev mode, per-node identities and a shared SE key), `p1-unique` (p1 with a unique name per document instead of the 40-name pool, so an SE query matches exactly one document: it separates the searchable-encryption first-responder misses from the "many documents per name" query shape; run 307 remains the colliding-name result) or `p2-acp` (User under a local ACP policy with owner/reader identities, see "ACP profile"). | +| `--create-nodes 0,1` | all nodes | Node indices that receive create ops (0,1 Rust; 2,3 Go); other ops still go to any node. Recorded in the manifest. | +| `--ops N` | 200 | Ops to plan and execute. | +| `--secs S` | none | Wall deadline; stops the workload first if hit. | +| `--rate R` | 20 | Profile rate, ops/s mesh-wide, and the virtual clock (`virtual_ts = index / rate`). ~3 is sustainable for 1R+1G on a MacBook. | +| `--churn` | off | Enable the seeded restart / crash-kill / graceful-leave schedule. | +| `--churn-spacing S` | 120 | Mean seconds between events and per-node cooldown. | +| `--grace S` | 120 | Mismatches younger than this, or within this long after a node came back, are in-flight sync, not divergence. Covers two failed pushes on the runtimes' 30/60/120s retry ladder. | +| `--settle S` | 120 | After the workload, keep checking this long for an eligible clear check before the final sweep. | +| `--min-settle S` | 0 | Settle at least this long even once the mesh is clear. The default ends the settle on the first clear check, often seconds after the last op, so a run has no sample of an idle mesh: `du` and RSS stop at the load. Set it to get an idle tail; the meter samples throughout. | +| `--ceiling-mb MB` | 122880 | Disk ceiling over all node data dirs; hard stop at 95%. | +| `--floor-rate R` | 0.5 | The governor never throttles below this. | +| `--meter-secs S` | 60 | du / RSS sampling and governor interval. | +| `--control` | off | Positive control: a `Control` collection replicated rust-0 -> go-0 only, written on go-0, must produce divergences on the pairs that predicts and nothing on `Users`. | +| `--retry-intervals 5,10,20,40` | runtime default | Both runtimes' `--replicator-retry-intervals`, on both backends; recorded in the manifest since it changes the system under test. Without it an outage longer than the first 30 s rung is timed by the sender's next dial, not by replication, so recovery numbers are ladder-confounded and the two runtimes are not comparable across that gap. | +| `--node-env KEY=VALUE` | none | Repeatable. Set on every node, both backends (docker `-e`, process inherited from the driver) and both runtimes, and recorded in `manifest.caps.node_env` so a run's log level is auditable. Only validation is the `=`. The Rust partition-tail lines (`dag_fetcher.rs` "Attempt stall budget exhausted", `swarm.rs` "Closing redundant connection", `retry.rs` "Activated durable push markers") are DEBUG, so no existing run contains them: `--node-env RUST_LOG=debug`. | +| `--no-subscribe` | off | Skip the collection subscribe (`p2p_collection_add`), leaving the replicators as the only delivery path. By default every node both subscribes to the collection topic and has a replicator to every other node, so gossip delivers whatever a replicator push loses and a broken push is invisible. Use it to measure the replicator alone. | +| `--sse-go` | off | Open subscriptions on Go nodes too (reproduces the Go memory growth). | +| `--nodes process\|docker` | process | `docker` runs the M2 six-node topology (rust-0, rust-1, go-0 on host A; rust-2, go-1, go-2 on host B) as containers on a `soak-` network, see "Docker backend". Recorded per node in the manifest (`backend`) and honoured by `replay`. | +| `--reuse-network` | off | Start even though a `soak-*` network is left over from an earlier run. | +| `--topology rg` | backend default | Mesh shape: `n` Rust nodes, then `m` Go nodes. Either count may be zero, so `4r0g` and `0r4g` are the single-runtime controls for a mixed run. Recorded in the manifest and honoured by `replay`. Without it each backend keeps the shape every published run used: the M2 six in containers, two of each as processes. Node **order is Rust first**, so `--create-nodes 0,1` means the first two Rust nodes at `2r2g` but the first two Go nodes at `0r4g`. | + +### Docker backend + +`--nodes docker` needs the images `soak-defra:8d8bb299f` (Rust) and +`soak-defradb:$DEFRA_GO_COMPAT_COMMIT` (Go) on the docker host, so the Go +image and the host `defradb` always name the same version; build the Go +image locally for that version, e.g. for v1.1.0, from a defradb clone at +the tag: `docker build --platform linux/arm64 -f tools/defradb.containerfile +--build-arg VERSION=v1.1.0 -t soak-defradb:v1.1.0 `. It +also needs `DEFRA_RUST_BINARY` and the Go `defradb` on `PATH` as before +(the driver's CLI calls run on the host against each container's published +API port), and the docker CLI pointed at the host, e.g. +`DOCKER_CONTEXT=orbstack`. The encrypted and ACP profiles refuse the +backend; plaintext profiles (`p0-crud`, `p0-size`, `p0-size-256`, `p0-size-128k`, +`p0-index`, `p3-relation`) run in containers, but 128 KB payloads have never +been exercised through the container path. + +```sh +export DOCKER_CONTEXT=orbstack +cargo run -p soak -- run --nodes docker --seed 611 --ops 300 --rate 3 +``` + +Each container mounts `/target/docker/` at `/data`, so the +node data stay in the artifact and `docker logs` are flushed to +`/logs/{stdout,stderr}.log` before every log rotation and before +teardown. The run removes its containers and network at the end, on error +too. A run that +was killed leaves them behind, and the next `soak run` refuses to start +while any `soak-*` network exists: remove them (`docker rm -f $(docker ps +-aq --filter name=soak-)`, then `docker network rm soak-`) or pass +`--reuse-network`. + +## Artifact + +``` +runs/-/ + manifest.json seed, profile, ops, nodes (store, peer id, backend, image, ip, host), both binaries' version + JSON, churn config + planned schedule, caps; at the end ops_executed, + stopped_by (ops | secs | budget | until_op) and the checker totals + ops.jsonl one record per executed op + topology.jsonl churn events as executed: down/up per event, planned vs actual + virtual time, wall time, peer id after recovery + checks.jsonl every checker pass: status, mismatch/pending/confirmed counts, eligibility + divergences.jsonl confirmed divergences (see below), with the pair and per-doc tags + final_sweep.jsonl every mismatch of the final full sweep, confirmed or not, with its tag + lag.jsonl convergence lag samples per create, by directed pair; source sse or poll + du.jsonl rss.jsonl budget.jsonl meter samples and governor decisions + profile.json/.md the per-runtime behaviour profile + target/e2e//{rust-0,rust-1,go-0,go-1}/{data,logs} node data dirs and stdout/stderr + (logs rotated to *.before-event-N before a restart) + target/docker//{data,logs} the same for --nodes docker +``` + +## How it works + +**Generator (axis 1).** `Profile::p0_crud` is a weight table (create 30, +update 40, delete 5, query 25, ~1.2 KiB docs). The op stream is a pure +function of `(seed, profile, node count)`: execution outcomes never feed +back. Update/delete victims are ledger *slots* in creation order; the +executor maps slots to the docIDs it learned from `add_X` replies. A victim +whose create failed (its node was down) is logged as `skipped`, not as an +error. An update or delete whose docID the target node does not hold yet +(replication lag) gets an empty reply from both runtimes; it is logged as +failed with `no doc matched on this node`. Floats are generated with at most 8 significant digits, under the +15-digit roundtrip ceiling. + +**Executor.** Every op is one HTTP GraphQL POST to its target node. The +executor is sequential, so throughput is capped at 1 / (average latency); +Rust writes take ~250 ms on regolith, so ~3 ops/s is the practical ceiling +for one Rust node today. + +**Checker.** One task for the whole mesh. Every `interval` (10 s): each +collection's docID set is fetched once per node and diffed for every pair +(M1); head CIDs via alias-batched `_commits(docID: ..., depth: 1)` are +fetched once per node over docs touched since the last check plus a cold +sample of 50 and diffed per pair (M3). For `p1-encrypted` the same targets +are also read as plaintext (`filter: {_docID: {_in: [..]}}`) per node and +compared per pair (M5); a null or empty encrypted field on one side only is +recorded as `undecryptable_on`. A mismatch becomes a divergence +record only after it persisted across 3 consecutive checks spanning at +least `grace`; the record names the pair and carries the op range and node +down/up transitions since the last clear check. A pair is eligible when +both members are up and past `grace` since their last recovery; mismatches +on ineligible pairs are logged as `expected` and their pending state is +frozen, neither counted nor cleared. Each divergent doc gets a known-cause +tag when its last write happened while a pair member was down +(`write-during-outage`) or within 30 s of the writer's or a member's +recovery (`write-during-recovery`); the summary's alarm line is the +count of untagged docs. After the workload the +checker settles (checks until a fully clear, fully eligible check or +`--settle` runs out), then sweeps M3 over every shared doc. The summary reports records +written, how many are still present at the final sweep, and the final +sweep's own mismatch count, so a backlog that drains is distinguishable +from a real split. + +**Churner (axis 2).** The schedule is drawn at start from the topology +stream: mean spacing, per-node cooldown, kinds drawn uniformly (restart, +crash-kill, graceful leave, and partition on `--nodes docker`), down-time +in 5..30 s. Events fire on wall time from workload start (polled every +250 ms), and the schedule covers the shorter of the op budget and the +`--secs` deadline; manifests without a `churn.config.clock` replay on +virtual time (op progress) as they ran. `topology.jsonl` records planned and actual +clock time. `restart` is the harness's SIGTERM path (same ports, health gate); +`crash_kill` is SIGKILL, a wall-time pause, respawn, then a GraphQL health +poll; `graceful_leave` is the harness's stop (SIGTERM, ports held), a +wall-time pause, then its start on the same ports; `partition` is a docker +network disconnect, a wall-time pause, then connect (the API port goes +with the network, so the driver sees a crash-kill and the `rejoin` record +carries the re-read `p2p_addr`). Down-time is wall time so a stalled workload cannot leave a node +dead. The churner shares the driver task with the workload (the harness +restart future is not `Send`). + +**Meter and governor.** From the churner task (it holds the cluster): `du` +of each data dir and the process RSS. Amplification is the cumulative +mesh-wide bytes grown per executed op since the first sample. The governor +sets the op rate that would spend the remaining budget exactly by the +deadline (op-count runs derive one from the remaining ops at the profile +rate), clamped to `[floor, profile rate]`, and raises the hard stop at 95% +of the ceiling; the stop is evaluated per sample, so it can overshoot by one +interval's growth. + +**Subscriptions.** One `subscription { Users { _docID } }` is kept open per +Rust node over the GraphQL POST endpoint with `Accept: text/event-stream`, +reconnecting after a restart (`--sse-go` opens them on Go nodes too). Arrivals mark docs recent for M3, give lag +samples at event time (`source: sse`), and 5 s of silence after any event +triggers a check ahead of the clock. Go's subscription fires only for the +node's own mutations, not for remote merges, so event-time lag exists only +into Rust nodes and quiescence triggering is partial; the clock is the +trigger that matters. + +**Replay contract.** What `compare` checks: for each op index the planned +fields (`virtual_ts_ms`, `node`, `kind`, `collection`) are identical, the +churn schedules are identical, and docIDs agree wherever both runs learned +one (content-addressed docIDs make payload regeneration exact). Outcomes, +error text, latency and wall time are not part of the contract: crash +down-windows are wall time, so ops at their edges may fail in one run and +succeed in the other. + +## Reading a run + +- `divergence records: N, still present at final sweep: K` with `final + sweep: 0 mismatches (eligible)` means every confirmed mismatch healed: + lag, not a split. Look at `checks.jsonl` for the backlog shape and at + `lag.jsonl` for the direction. +- A non-zero final sweep with `eligible` is the headline; `final_sweep.jsonl` + names the docs, pairs, sides and tags, and `profile.md` classifies each by + the last write versus the outage windows. +- `divergent docs: N tagged, M UNTAGGED` is the alarm line: tagged docs are + the known write-during-outage loss; untagged ones need a look. +- Counts are record-doc slots unless the line says `unique`: a doc missing on + one node is one row per pair that node belongs to, so `1277 record-doc + slots` can be `259 unique docs`. The `final sweep:` line and the `loss` + tables are unique documents. +- `sampled_pending` is what that pass happened to look at (recent docs plus a + cold sample of 50), not the size of the backlog. +- Convergence lag is in ms and split by `source`: `poll` is bounded below by + the 10 s checker interval, `sse` is event time. The `*->rust` / `*->go` + rows are by receiving runtime. Creates that never arrived have no sample at + all and are listed as unseen under the table, not as a fast percentile. +- `bytes_grown_per_mesh_write` divides one node's growth by the whole mesh's + successful writes: it is replication amplification, not that node's writes. +- On the docker backend the memory heading says `docker stats MemUsage`; the + values are whatever `docker stats` printed for the container, not `ps` RSS + of one process. +- The `loss` tables count creates made by some other node while a node was + down and still missing on it at the final sweep. `[down,up]` is the strict + window; `[down,up+30s]` adds the recovery window, since a node answers + GraphQL before its replicator link is back. +- `NOT eligible` on the final sweep means a node was down or in grace at the + end; lengthen `--settle`. + +Known runtime behaviours met while building M0 (Rust `ba6dac661`, Go +`53f0e76a3`, macOS arm64): + +- Rust HTTP writes take ~250 ms (create/update/delete) against Go's ~3 ms; + debug and release builds alike. Queries are ~1-3 ms on both. +- Disk: ~8.4 KB per write op on rust-0/regolith vs ~5.8 KB on go-0/badger, + engine-inclusive. Max RSS ~107 MB vs ~238 MB. +- Go -> Rust pushes fail under load (Rust's DAG block fetches from Go time + out and back off; Rust rejects re-pushes of an in-flight CID) and Go + retries on its 30/60/120/240 s ladder, so single docs took 20 s to 7 min + to converge at 3 ops/s. That is why `--grace` defaults to 120 s. +- Without a file keyring a Go node comes back from every restart as a new + peer ID and the Rust replicator never reconnects to it; the driver uses + `TestClusterBuilder::with_file_keyring()`. +- A create, update or delete made on either node while its peer is + crash-killed, or within seconds of a node's recovery, is never replicated + afterwards (the write-during-outage tag). Unchanged by the retry ladder. +- With four nodes in a full mesh, every direction converges in 4-8 s median; + the 55 s Go-to-Rust median of a single pair does not appear. +- Go's GraphQL subscriptions do not fire for remote merges, and a Go node + with one open grows by roughly 200 MB of resident memory per minute at + 3 ops/s until it restarts (7.8 GB after 30 minutes); Rust nodes do not. + That is why the driver subscribes on Rust nodes only. + +### Topology + +`--topology rg` sets the mesh size and runtime mix. Nodes are named +`rust-0..rust-` then `go-0..go-`, and each runtime's nodes alternate +between the two partition sides so neither side is single-runtime. + +```sh +soak run --topology 2r2g # two of each, the published process shape +soak run --topology 4r0g # all-Rust control +soak run --topology 0r4g # all-Go control +soak run --topology 6r2g # asymmetric +``` + +A single-runtime mesh skips the steps that need a node of each kind and says +so: the `--control` wiring, and the `p2-acp` token probe against the absent +runtime. `p2-acp` itself needs a Rust node, because it mints identities with +the Rust CLI. + +One cost grows sharply with the mesh, and it is not the checker. The checker +reads each node once per pass and compares the unordered pairs in memory, so +its HTTP cost is linear in nodes. `profile.md`, though, prints one row per +*directed* pair: 12 rows at four nodes, 30 at six, 380 at twenty, which stops +being readable well before that. Node memory is the real ceiling. Size the run +to the host. + +## Management channel + +`soak manage` is a minutes-long evaluation of `POST /api/v0/p2p/manage`: the +caller hits one Rust node's HTTP API (the relay) with a JWT whose `aud` is +the target's peer id, and the relay carries the op over P2P to the target, +which authorizes the actor against NAC before applying it. + +```sh +DEFRA_RUST_BINARY=/target/debug/defra \ + soak manage --topology 2r0g --cases R2,A2,S1 --out runs/manage-1 +``` + +The cluster is the `run` mesh with NAC enabled (`--node-acp-enable` and a +startup identity, which is the NAC owner and the HTTP courier at every +relay). Every node gets the `User` schema, peer connections and a replicator +to every other node, all as the owner. Three actors are generated and granted +on every node through `acp node relationship add`: `admin` (the `admin` +relation), `operator` (`add-p2p-collection` and `list-p2p-replicator` only), +`outsider` (nothing). + +`--transport iroh` runs the same table on the iroh transport; the binary +`DEFRA_RUST_BINARY` names must then be built with `--features iroh`. The +transport is recorded in `manifest.json`, `summary.json` and the `cases.md` +heading, and the bounds cases size against the transport's request bound +(`bounds.rs`). + +A topology with Go nodes (`--topology 2r2g`, libp2p only: Go does not speak +iroh) puts them in the same mesh as replication peers, under the same NAC +setup and owner, with the schema minus `@immutable` (Go lacks the directive; +it does not enter the collection id). Go has no manage protocol, so a Go +node is never a relay or a target; the cases still address nodes `0..rust`. +`H1` runs the cases two Rust nodes can host (R2, A1, A2, S1, S3, S4), each +its own row, and its own row says whether every Go node converged on the +source's documents after S3 and S4 and kept the replicator set the mesh gave +it. Without Go nodes H1 skips. `manifest.json` records each node's runtime. + +Cases live by group in +`src/manage/{routing,authz,state,bounds,partition,hybrid}.rs`, the table and +runner in `cases.rs`; each restores what it changed. `--cases` +runs the named cases in the order given; the default is every case but B3 in +table order, with the bounds group last so a target they wedge cannot poison +the rest. B3 locates the transport's request size bound by bisection and runs +under `--locate-size-bound` (or by name) on its own. Before each case the +runner sends the cheapest admin query to every node the case uses and to +every Go node; a node that no longer answers makes the case `Infra`, naming +the last case that used it. A case whose topology requirement the mesh cannot host is skipped, not +failed. Outcomes: `Pass`, `Fail { expected, got }`, `Skip { reason }`, +`Infra { error }` (a harness fault, never a product finding). `--out` receives `manifest.json` (nodes, peer ids, actors in +cleartext like `run`), `summary.json` (per case: outcome, notes a case +recorded, every relayed op with status and latency, and the target's list for +that op's family after each mutate) and `cases.md`. `--docker` is not +supported yet, so P1 (partition) skips. + +## Not yet + +A second machine, per-pair partitions, link degradation (tc/netem), +relations / secondary indexes / lens, node-internal telemetry (otel), a +concurrent executor, M1 sweep scoping, tag rules for anything but the +outage loss. diff --git a/crates/soak/docker/defra.Dockerfile b/crates/soak/docker/defra.Dockerfile new file mode 100644 index 0000000..ec1611a --- /dev/null +++ b/crates/soak/docker/defra.Dockerfile @@ -0,0 +1,15 @@ +# Rust DefraDB image for the soak (M2). Same as defradb.rs/Dockerfile at 8d8bb299f plus +# libdbus-1-dev / libdbus-1-3, which the keyring crate's libdbus-sys needs on Linux and the +# upstream file omits (build fails at libdbus-sys v0.2.7). Build with the defradb.rs checkout +# as the context: docker build -f crates/soak/docker/defra.Dockerfile -t soak-defra: +FROM rust:1.93-bookworm AS builder +WORKDIR /build +COPY . . +RUN apt-get update && apt-get install -y libssl-dev pkg-config protobuf-compiler libdbus-1-dev +RUN cargo build --release -p cli + +FROM debian:bookworm-slim +RUN apt-get update && apt-get install -y ca-certificates libssl3 libdbus-1-3 && rm -rf /var/lib/apt/lists/* +COPY --from=builder /build/target/release/defra /usr/local/bin/defra +EXPOSE 9161 9171 9181 +ENTRYPOINT ["defra"] diff --git a/crates/soak/src/auth.rs b/crates/soak/src/auth.rs new file mode 100644 index 0000000..6d93140 --- /dev/null +++ b/crates/soak/src/auth.rs @@ -0,0 +1,192 @@ +//! Bearer tokens for identity-scoped HTTP requests (copy of the integration +//! tests' `auth_token`, tools/integration-test/tests/acp/events_sse.rs). + +use std::collections::HashMap; +use std::time::{Duration, Instant}; + +use eyre::{Result, WrapErr}; +use serde::{Deserialize, Serialize}; + +use crate::generator::Actor; + +#[derive(Clone, Debug, Serialize, Deserialize)] +pub struct Identity { + pub key_hex: String, + pub did: String, +} + +#[derive(Clone, Debug, Serialize, Deserialize)] +pub struct Identities { + pub owner: Identity, + pub reader: Identity, +} + +/// A 15-minute bearer token for `key_hex`, audience = the API host with port. +pub fn auth_token(key_hex: &str, api_url: &str) -> Result { + token(key_hex, host_port(api_url)) +} + +/// A 15-minute actor token for the P2P management channel, audience = the +/// target node's peer id (copy of the integration tests' `mint_manage_token`, +/// tools/integration-test/tests/manage_relay_common.rs). +pub fn manage_token(key_hex: &str, target_peer_id: &str) -> Result { + token(key_hex, target_peer_id) +} + +fn token(key_hex: &str, audience: &str) -> Result { + let key = hex::decode(key_hex).wrap_err("identity hex")?; + let key_type = match key.len() { + 32 => crypto::KeyType::Secp256k1, + 64 => crypto::KeyType::Ed25519, + n => eyre::bail!("unsupported identity length {n}"), + }; + let raw = identity::RawIdentity::from_bytes(key_type, &key).wrap_err("raw identity")?; + let token = identity::new_token( + &raw, + Duration::from_secs(15 * 60), + Some(audience.to_string()), + None, + ) + .wrap_err("mint token")?; + String::from_utf8(token).wrap_err("token utf-8") +} + +/// `api_url` without its scheme: the token audience and the CLI `--url` both +/// want `host:port` (the CLI prepends its own scheme). +pub fn host_port(api_url: &str) -> &str { + api_url + .strip_prefix("https://") + .or_else(|| api_url.strip_prefix("http://")) + .unwrap_or(api_url) +} + +/// Tokens per (actor, node url), re-minted after 10 minutes (they expire at 15). +pub struct TokenCache { + ids: Identities, + tokens: HashMap<(Actor, String), (String, Instant)>, +} + +impl TokenCache { + pub fn new(ids: Identities) -> Self { + Self { + ids, + tokens: HashMap::new(), + } + } + + pub fn identities(&self) -> &Identities { + &self.ids + } + + pub fn bearer(&mut self, actor: Actor, api_url: &str) -> Result> { + let key = match actor { + Actor::Anon => return Ok(None), + Actor::Owner => &self.ids.owner.key_hex, + Actor::Reader => &self.ids.reader.key_hex, + }; + let k = (actor, api_url.to_string()); + if let Some((tok, minted)) = self.tokens.get(&k) { + if minted.elapsed() < Duration::from_secs(600) { + return Ok(Some(tok.clone())); + } + } + let tok = auth_token(key, api_url)?; + self.tokens.insert(k, (tok.clone(), Instant::now())); + Ok(Some(tok)) + } +} + +#[cfg(test)] +mod tests { + use super::*; + const KEY: &str = "1111111111111111111111111111111111111111111111111111111111111111"; + + #[test] + fn token_is_a_jwt_with_host_port_audience() { + let tok = auth_token(KEY, "http://127.0.0.1:55110").unwrap(); + let parts: Vec<&str> = tok.split('.').collect(); + assert_eq!(parts.len(), 3, "{tok}"); + let claims = base64_url_decode(parts[1]); + assert!(claims.contains("\"aud\":[\"127.0.0.1:55110\"]"), "{claims}"); + } + + #[test] + fn manage_token_audience_is_the_target_peer_id() { + let tok = manage_token(KEY, "12D3KooWTarget").unwrap(); + let parts: Vec<&str> = tok.split('.').collect(); + assert_eq!(parts.len(), 3, "{tok}"); + let claims = base64_url_decode(parts[1]); + assert!(claims.contains("\"aud\":[\"12D3KooWTarget\"]"), "{claims}"); + } + + #[test] + fn host_port_strips_scheme() { + assert_eq!(host_port("http://127.0.0.1:5"), "127.0.0.1:5"); + assert_eq!(host_port("127.0.0.1:5"), "127.0.0.1:5"); + } + + #[test] + fn cache_returns_none_for_anon_and_reuses_tokens() { + let ids = Identities { + owner: Identity { + key_hex: KEY.into(), + did: "did:key:owner".into(), + }, + reader: Identity { + key_hex: KEY.into(), + did: "did:key:reader".into(), + }, + }; + let mut c = TokenCache::new(ids); + assert_eq!( + c.bearer(crate::generator::Actor::Anon, "http://a:1") + .unwrap(), + None + ); + let t1 = c + .bearer(crate::generator::Actor::Owner, "http://a:1") + .unwrap() + .unwrap(); + let t2 = c + .bearer(crate::generator::Actor::Owner, "http://a:1") + .unwrap() + .unwrap(); + assert_eq!(t1, t2); + assert_ne!( + t1, + c.bearer(crate::generator::Actor::Owner, "http://b:2") + .unwrap() + .unwrap() + ); + } + + fn base64_url_decode(s: &str) -> String { + let mut s = s.replace('-', "+").replace('_', "/"); + while !s.len().is_multiple_of(4) { + s.push('='); + } + // minimal decoder to avoid a base64 dep in tests + let table: Vec = (b'A'..=b'Z') + .chain(b'a'..=b'z') + .chain(b'0'..=b'9') + .chain([b'+', b'/']) + .collect(); + let mut out = Vec::new(); + let mut buf = 0u32; + let mut bits = 0; + for ch in s.bytes() { + if ch == b'=' { + break; + } + let v = table.iter().position(|t| *t == ch).unwrap() as u32; + buf = (buf << 6) | v; + bits += 6; + if bits >= 8 { + bits -= 8; + out.push((buf >> bits) as u8); + buf &= (1 << bits) - 1; + } + } + String::from_utf8_lossy(&out).into_owned() + } +} diff --git a/crates/soak/src/checker.rs b/crates/soak/src/checker.rs new file mode 100644 index 0000000..b3a204a --- /dev/null +++ b/crates/soak/src/checker.rs @@ -0,0 +1,1123 @@ +//! Convergence checker for an N-node mesh. +//! +//! Every check: the docID set of each collection is fetched once per node +//! (M1), diffed for every pair; head CIDs (`_commits(docID, depth: 1)`) are +//! fetched once per node over the docs touched since the last check plus a +//! cold sample (M3) and diffed per pair; the shutdown check sweeps every +//! shared doc. A pair is eligible when both members are up and past +//! `grace` since their last recovery; mismatches on ineligible pairs are +//! expected and their pending state is frozen. Eligible mismatches pass +//! through [`Confirmer`] before they become records in `divergences.jsonl`; +//! every check appends a line to `checks.jsonl`. +//! +//! ponytail: the M1 sweep is O(docs) per node per check, fine at M1 scale +//! (<20k docs); add recent-window scoping when a check exceeds ~1s. + +use std::collections::{HashMap, HashSet}; +use std::fs::File; +use std::io::{BufWriter, Write}; +use std::path::Path; +use std::sync::atomic::{AtomicU64, Ordering}; +use std::sync::Arc; +use std::time::{Duration, Instant}; + +use eyre::{eyre, Result, WrapErr}; +use rand::{rngs::StdRng, seq::SliceRandom, SeedableRng}; +use serde_json::{json, Value}; +use tokio::sync::{mpsc, oneshot}; + +use crate::auth::{Identities, TokenCache}; +use crate::churn::Transition; +use crate::confirm::{Confirmer, Key}; +use crate::executor::{gql_as, http_client, now_ms}; +use crate::generator::Actor; +use crate::sse::Arrival; +use crate::tags::{tag, Outage, Tag}; + +/// After the last subscription event, this much silence triggers a check +/// ahead of the clock. +const QUIET: Duration = Duration::from_secs(5); + +/// A confirmed mismatch: key, detail, checks it persisted across. +type Confirmed = (Key, Value, u32); + +/// A doc the workload just wrote; feeds M3 scoping and, for creates, the +/// convergence-lag samples. +#[derive(Debug)] +pub struct Touch { + pub doc_id: String, + pub node: String, + pub wall_ts_ms: u64, + pub create: bool, + /// Creates on ACP profiles: whether the doc got a policy (owner-created). + /// None on p0/p1 and on non-creates. + pub protected: Option, +} + +/// Forget a create still unseen somewhere after this long; by then it is a +/// divergence, not a lag sample. +const LAG_TTL_MS: u64 = 900_000; + +/// docID -> one value per configured encrypted field (None = null/absent). +pub type FieldMap = HashMap>>; + +/// M5: a doc both nodes hold must read the same plaintext on both. A null +/// or empty value on one side only is "replicated but undecryptable". +pub fn m5_compare( + a: &FieldMap, + b: &FieldMap, + node_a: &str, + node_b: &str, + targets: &[String], + fields: &[String], +) -> Vec<(String, Value)> { + let mut out = Vec::new(); + for id in targets { + let (Some(x), Some(y)) = (a.get(id), b.get(id)) else { + continue; + }; + for (i, field) in fields.iter().enumerate() { + let (vx, vy) = (x.get(i).cloned().flatten(), y.get(i).cloned().flatten()); + let empty = |v: &Option| v.as_deref().is_none_or(str::is_empty); + let detail = match (empty(&vx), empty(&vy)) { + (true, true) => continue, + (true, false) => json!({ "undecryptable_on": node_a, "field": field }), + (false, true) => json!({ "undecryptable_on": node_b, "field": field }), + (false, false) if vx != vy => { + json!({ "field": field, "a": short(&vx), "b": short(&vy) }) + } + _ => continue, + }; + out.push((id.clone(), detail)); + break; + } + } + out +} + +/// First 8 chars: enough to see two values differ without logging secrets. +/// Settle ends once the floor has been served and either the mesh is clear +/// or the budget is spent. A zero floor is the historical early exit. +fn settle_done(elapsed: Duration, min_settle: Duration, settle: Duration, clear: bool) -> bool { + elapsed >= min_settle && (clear || elapsed >= settle) +} + +fn short(v: &Option) -> String { + v.as_deref().unwrap_or("").chars().take(8).collect() +} + +/// docID -> visible to [owner, reader, anon]. +pub type ViewMap = HashMap; +const VIEWERS: [&str; 3] = ["owner", "reader", "anon"]; +const VIEWER_ACTORS: [Actor; 3] = [Actor::Owner, Actor::Reader, Actor::Anon]; + +/// M6: the three views of a protected doc must agree across a pair unless +/// exactly one side is the doc's origin node (local ACP gates only there). +pub fn m6_compare( + a: &ViewMap, + b: &ViewMap, + names: &[String], + idx_a: usize, + idx_b: usize, + origin: &HashMap, + targets: &[String], +) -> (Vec<(String, Value)>, usize) { + let mut out = Vec::new(); + let mut by_design = 0; + for id in targets { + let (Some(x), Some(y), Some(o)) = (a.get(id), b.get(id), origin.get(id)) else { + continue; + }; + if (*o == idx_a) != (*o == idx_b) { + by_design += 1; + continue; + } + if let Some(i) = (0..3).find(|i| x[*i] != y[*i]) { + out.push(( + id.clone(), + json!({ "viewer": VIEWERS[i], "a": x[i], "b": y[i], "origin": names[*o] }), + )); + } + } + (out, by_design) +} + +pub struct CheckerConfig { + pub interval: Duration, + /// A mismatch younger than this is "sync in flight", never a divergence. + /// Both runtimes retry a failed push after 30s then 60s, so a doc whose + /// push failed twice lands at ~90s; the default covers that. + pub grace: Duration, + pub confirmations: u32, + /// Untouched shared docs to head-check per interval check. + pub cold_sample: usize, + /// `_commits` aliases per POST. + pub batch: usize, + /// After the workload stops, keep checking this long for a clear check + /// before the final sweep, so in-flight sync is not read as divergence. + pub settle: Duration, + /// Settle for at least this long even if the mesh is already clear, so a + /// run that wants an idle sample gets one after the load stops. + pub min_settle: Duration, + /// Encrypted fields to compare as plaintext (M5); empty = off. + pub encrypted_fields: Vec, + /// Owner/reader identities for the access-parity views (M6); None = off. + pub acp: Option, +} + +impl Default for CheckerConfig { + fn default() -> Self { + Self { + interval: Duration::from_secs(10), + grace: Duration::from_secs(120), + confirmations: 3, + cold_sample: 50, + batch: 100, + settle: Duration::from_secs(120), + min_settle: Duration::ZERO, + encrypted_fields: Vec::new(), + acp: None, + } + } +} + +#[derive(Debug, Default)] +pub struct Summary { + pub checks: u64, + /// Checks in which at least one node did not answer. + pub unreachable: u64, + /// Divergence records written. + pub divergences: u64, + /// Confirmed mismatches still present at the final sweep. + pub unresolved: usize, + /// Mismatches in the final full sweep over every pair. + pub final_mismatches: usize, + /// Whether every pair was eligible and every node reachable at the sweep. + pub final_eligible: bool, + /// Divergent docs in records with a known-cause tag, and without one. + pub tagged_docs: u64, + pub untagged_docs: u64, + /// Subscription events received per node. + pub sse_events: Vec, + /// Checks triggered by quiescence rather than the clock. + pub quiet_checks: u64, + /// Docs whose plaintext was compared on both sides of a pair (M5), summed + /// over checks; zero means M5 never compared anything. + pub m5_docs: u64, + /// Protected docs whose three views were compared across a pair (M6), + /// and those skipped because one side was the doc's origin node. + pub m6_docs: u64, + pub m6_by_design: u64, +} + +/// What one comparison pass found. +struct Compared { + /// Mismatches on eligible pairs. + mismatches: Vec<(Key, Value)>, + /// All mismatches, eligible or not, with their pair (for the final sweep). + all: Vec<(Key, Value)>, + m3_docs: usize, + /// Docs present in both maps of a pair and so actually compared by M5. + m5_docs: usize, + m6_docs: usize, + m6_by_design: usize, + unreachable: Vec, +} + +pub struct Checker { + http: reqwest::Client, + /// (name, api_url) per node index. + nodes: Vec<(String, String)>, + /// Unordered pairs (a < b) and their key string `"|"`. + pairs: Vec<(usize, usize, String)>, + collections: Vec, + cfg: CheckerConfig, + /// Bearer tokens for the M6 views; None off ACP profiles. + tokens: Option, + rng: StdRng, + confirmer: Confirmer, + run_id: String, + seed: u64, + /// Ops issued so far, kept current by the driver. + op_index: Arc, + /// Op index at the last fully clear check; the record's event window + /// starts here. + last_clear_op: u64, + /// Nodes currently down per the churner. + down: HashSet, + /// Per node: mismatches before this instant are expected (it came back + /// less than `grace` ago). + eligible_at: Vec, + /// Node down/up transitions since the last clear check. + window_events: Vec, + /// Every outage so far, for the known-cause tags. + outages: Vec, + /// Last successful write per doc: (node, wall ms). + last_write: HashMap, + /// Per created doc: (origin node index, protected), from creates. + doc_meta: HashMap, + checks: BufWriter, + divergences: BufWriter, + /// Every mismatch of the final sweep, confirmed or not. + final_sweep: BufWriter, + /// Creates not yet seen everywhere: doc -> (origin, wall ms, nodes seen on). + awaiting: HashMap)>, + lag: BufWriter, + summary: Summary, +} + +impl Checker { + #[allow(clippy::too_many_arguments)] + pub fn new( + nodes: Vec<(String, String)>, + collections: Vec, + cfg: CheckerConfig, + seed: u64, + run_id: String, + op_index: Arc, + run_dir: &Path, + ) -> Result { + let open = |name: &str| -> Result> { + let path = run_dir.join(name); + Ok(BufWriter::new(File::create(&path).wrap_err_with(|| { + format!("creating {}", path.display()) + })?)) + }; + let mut pairs = Vec::new(); + for a in 0..nodes.len() { + for b in a + 1..nodes.len() { + pairs.push((a, b, format!("{}|{}", nodes[a].0, nodes[b].0))); + } + } + Ok(Self { + http: http_client(Duration::from_secs(30)), + eligible_at: vec![Instant::now(); nodes.len()], + nodes, + pairs, + collections, + confirmer: Confirmer::new(cfg.confirmations, cfg.grace), + tokens: cfg.acp.clone().map(TokenCache::new), + cfg, + rng: StdRng::seed_from_u64(seed), + run_id, + seed, + op_index, + last_clear_op: 0, + down: HashSet::new(), + window_events: Vec::new(), + outages: Vec::new(), + last_write: HashMap::new(), + doc_meta: HashMap::new(), + checks: open("checks.jsonl")?, + divergences: open("divergences.jsonl")?, + final_sweep: open("final_sweep.jsonl")?, + awaiting: HashMap::new(), + lag: open("lag.jsonl")?, + summary: Summary::default(), + }) + } + + /// Check every `interval` until `stop` fires, then settle and run the + /// full sweep. `touched` feeds docIDs the workload wrote since the last + /// check; `transitions` feeds node down/up events from the churner. + pub async fn run( + mut self, + mut touched: mpsc::UnboundedReceiver, + mut transitions: mpsc::UnboundedReceiver, + mut arrivals: mpsc::UnboundedReceiver, + mut stop: oneshot::Receiver<()>, + ) -> Result { + self.summary.sse_events = vec![0; self.nodes.len()]; + let mut tick = tokio::time::interval(self.cfg.interval); + tick.tick().await; // the immediate first tick + let mut recent = HashSet::new(); + let mut quiet_until: Option = None; + loop { + let quiet = async { + match quiet_until { + Some(t) => tokio::time::sleep_until(t).await, + None => std::future::pending::<()>().await, + } + }; + tokio::select! { + _ = tick.tick() => { + while let Ok(t) = touched.try_recv() { + self.note(t, &mut recent); + } + self.check(std::mem::take(&mut recent), &mut transitions, false) + .await?; + } + Some(a) = arrivals.recv() => { + self.arrival(a, &mut recent)?; + quiet_until = Some(tokio::time::Instant::now() + QUIET); + } + _ = quiet => { + quiet_until = None; + self.summary.quiet_checks += 1; + while let Ok(t) = touched.try_recv() { + self.note(t, &mut recent); + } + self.check(std::mem::take(&mut recent), &mut transitions, false) + .await?; + tick.reset(); + } + _ = &mut stop => { + // Settle: keep checking until a fully clear, eligible + // check or the budget runs out, then sweep everything. + let started = Instant::now(); + loop { + while let Ok(t) = touched.try_recv() { + self.note(t, &mut recent); + } + while let Ok(a) = arrivals.try_recv() { + self.arrival(a, &mut recent)?; + } + let clear = self + .check(std::mem::take(&mut recent), &mut transitions, false) + .await?; + if settle_done(started.elapsed(), self.cfg.min_settle, self.cfg.settle, clear) + { + break; + } + tokio::time::sleep(self.cfg.interval).await; + } + self.check(HashSet::new(), &mut transitions, true).await?; + self.summary.unresolved = self.confirmer.unresolved(); + return Ok(self.summary); + } + } + } + } + + fn note(&mut self, t: Touch, recent: &mut HashSet) { + if let Some(node) = self.nodes.iter().position(|(n, _)| *n == t.node) { + if t.create { + self.awaiting + .insert(t.doc_id.clone(), (node, t.wall_ts_ms, HashSet::new())); + if let Some(protected) = t.protected { + self.doc_meta.insert(t.doc_id.clone(), (node, protected)); + } + } + self.last_write + .insert(t.doc_id.clone(), (node, t.wall_ts_ms)); + } + recent.insert(t.doc_id); + } + + /// A subscription event: the doc is recent for M3, and if it is a create + /// still awaited on that node, a lag sample with event-time resolution. + fn arrival(&mut self, a: Arrival, recent: &mut HashSet) -> Result<()> { + if let Some(c) = self.summary.sse_events.get_mut(a.node) { + *c += 1; + } + let total = self.nodes.len(); + let mut line = None; + if let Some((origin, created, seen)) = self.awaiting.get_mut(&a.doc_id) { + if *origin != a.node && seen.insert(a.node) { + line = Some(json!({ + "wall_ts_ms": a.wall_ts_ms, "op_index": self.op_index.load(Ordering::Relaxed), + "doc_id": a.doc_id, "from": self.nodes[*origin].0, "to": self.nodes[a.node].0, + "lag_ms": a.wall_ts_ms.saturating_sub(*created), "source": "sse", + })); + if seen.len() >= total - 1 { + self.awaiting.remove(&a.doc_id); + } + } + } + if let Some(line) = line { + serde_json::to_writer(&mut self.lag, &line)?; + self.lag.write_all(b"\n")?; + self.lag.flush()?; + } + recent.insert(a.doc_id); + Ok(()) + } + + /// Known-cause tag for `doc` diverging on the pair `key`. + fn tag_for(&self, key: &str, doc: &str) -> Option { + let (a, b, _) = self.pairs.iter().find(|(_, _, k)| k == key)?; + let (writer, wall) = self.last_write.get(doc)?; + tag(*writer, *wall, (*a, *b), &self.outages) + } + + /// One check; returns whether it was fully clear (no mismatch on any + /// eligible pair, every pair eligible, every node reachable). `Err` only + /// for log I/O. A node that does not answer is treated as ineligible for + /// this check, so its pairs' pending state is frozen, not cleared. + async fn check( + &mut self, + recent: HashSet, + transitions: &mut mpsc::UnboundedReceiver, + full: bool, + ) -> Result { + let now = Instant::now(); + while let Ok(t) = transitions.try_recv() { + if t.up { + self.down.remove(&t.node); + if let Some(e) = self.eligible_at.get_mut(t.node) { + *e = now + self.cfg.grace; + } + if let Some(o) = self + .outages + .iter_mut() + .rev() + .find(|o| o.node == t.node && o.up_ms.is_none()) + { + o.up_ms = Some(t.wall_ts_ms); + } + } else { + self.down.insert(t.node); + self.outages.push(Outage { + node: t.node, + down_ms: t.wall_ts_ms, + up_ms: None, + }); + } + let node = self.nodes.get(t.node).map(|(n, _)| n.as_str()); + self.window_events.push(json!({ + "node": node, "up": t.up, "wall_ts_ms": t.wall_ts_ms, + })); + } + let mut node_ok: Vec = (0..self.nodes.len()) + .map(|n| !self.down.contains(&n) && now >= self.eligible_at[n]) + .collect(); + let op_index = self.op_index.load(Ordering::Relaxed); + self.summary.checks += 1; + + let compared = self.compare(&recent, full, &mut node_ok).await?; + self.summary.m5_docs += compared.m5_docs as u64; + self.summary.m6_docs += compared.m6_docs as u64; + self.summary.m6_by_design += compared.m6_by_design as u64; + let eligible_pairs = self + .pairs + .iter() + .filter(|(a, b, _)| node_ok[*a] && node_ok[*b]) + .count(); + let all_eligible = eligible_pairs == self.pairs.len(); + let n = compared.mismatches.len(); + let expected = compared.all.len() - n; + if full { + self.summary.final_mismatches = compared.all.len(); + self.summary.final_eligible = all_eligible && compared.unreachable.is_empty(); + for ((pair, col, mech, id), detail) in &compared.all { + let line = json!({ + "pair": pair, "collection": col, "mechanism": mech, "doc_id": id, + "detail": detail, "tag": self.tag_for(pair, id), + }); + serde_json::to_writer(&mut self.final_sweep, &line)?; + self.final_sweep.write_all(b"\n")?; + } + self.final_sweep.flush()?; + } + let ineligible: HashSet = self + .pairs + .iter() + .filter(|(a, b, _)| !(node_ok[*a] && node_ok[*b])) + .map(|(_, _, key)| key.clone()) + .collect(); + let confirmed = self + .confirmer + .observe(now, compared.mismatches, |k| ineligible.contains(&k.0)); + self.record(&confirmed, op_index)?; + let clear = n == 0 && all_eligible && compared.unreachable.is_empty(); + if clear { + self.last_clear_op = op_index; + self.window_events.clear(); + } + if !compared.unreachable.is_empty() { + self.summary.unreachable += 1; + } + let status = if clear { + "clear" + } else if !compared.unreachable.is_empty() { + "unreachable" + } else if n == 0 { + "expected" + } else { + "mismatch" + }; + let line = json!({ + "wall_ts_ms": now_ms(), "op_index": op_index, "full": full, + "eligible": all_eligible, "eligible_pairs": eligible_pairs, + "unreachable_nodes": compared.unreachable.iter().map(|n| &self.nodes[*n].0).collect::>(), + "status": status, "m3_docs": compared.m3_docs, "m5_docs": compared.m5_docs, "mismatches": n, "expected": expected, + "m6_docs": compared.m6_docs, "m6_by_design": compared.m6_by_design, + "pending": self.confirmer.pending(), "confirmed": confirmed.len(), + "duration_ms": now.elapsed().as_millis() as u64, + }); + serde_json::to_writer(&mut self.checks, &line)?; + self.checks.write_all(b"\n")?; + self.checks.flush()?; + Ok(clear) + } + + /// M1 + M3 (+ M5 for encrypted, M6 for ACP profiles) over all collections + /// and pairs. On ACP profiles the sweeps and fetches read as the owner so + /// protected documents are in scope; only the reader and anonymous views + /// of M6 use their own identities. + /// A node whose queries fail is added to `unreachable` and cleared in + /// `node_ok`. + async fn compare( + &mut self, + recent: &HashSet, + full: bool, + node_ok: &mut [bool], + ) -> Result { + let mut out = Compared { + mismatches: Vec::new(), + all: Vec::new(), + m3_docs: 0, + m5_docs: 0, + m6_docs: 0, + m6_by_design: 0, + unreachable: Vec::new(), + }; + for col in self.collections.clone() { + let mut ids: Vec>> = Vec::new(); + for (n, ok) in node_ok.iter_mut().enumerate() { + match self.doc_ids(n, &col).await { + Ok(set) => ids.push(Some(set)), + Err(_) => { + ids.push(None); + if !out.unreachable.contains(&n) { + out.unreachable.push(n); + } + *ok = false; + } + } + } + self.sample_lag(&ids)?; + + // M1 per pair. + for (a, b, key) in &self.pairs { + let (Some(sa), Some(sb)) = (&ids[*a], &ids[*b]) else { + continue; + }; + let eligible = node_ok[*a] && node_ok[*b]; + for (from, to, missing) in [(sa, sb, *b), (sb, sa, *a)] { + for id in from.difference(to) { + let entry = ( + (key.clone(), col.clone(), "M1", id.clone()), + json!({ "missing_on": self.nodes[missing].0 }), + ); + if eligible { + out.mismatches.push(entry.clone()); + } + out.all.push(entry); + } + } + } + + // M3: heads fetched once per node over the union of targets. + let mut held: HashMap<&String, usize> = HashMap::new(); + for set in ids.iter().flatten() { + for id in set { + *held.entry(id).or_default() += 1; + } + } + let universe: Vec<&String> = held + .iter() + .filter(|(_, n)| **n >= 2) + .map(|(id, _)| *id) + .collect(); + let targets: Vec = if full { + universe.iter().map(|s| s.to_string()).collect() + } else { + let (hot, cold): (Vec<&String>, Vec<&String>) = + universe.iter().partition(|id| recent.contains(**id)); + hot.into_iter() + .chain( + cold.choose_multiple(&mut self.rng, self.cfg.cold_sample) + .copied(), + ) + .cloned() + .collect() + }; + out.m3_docs += targets.len(); + let mut heads: Vec>>> = Vec::new(); + for (n, set) in ids.iter().enumerate() { + let Some(set) = set else { + heads.push(None); + continue; + }; + let mine: Vec = targets + .iter() + .filter(|t| set.contains(*t)) + .cloned() + .collect(); + let mut map = HashMap::new(); + let mut failed = false; + for chunk in mine.chunks(self.cfg.batch) { + match self.heads(n, chunk).await { + Ok(part) => map.extend(part), + Err(_) => { + failed = true; + break; + } + } + } + if failed { + heads.push(None); + if !out.unreachable.contains(&n) { + out.unreachable.push(n); + } + node_ok[n] = false; + } else { + heads.push(Some(map)); + } + } + for (a, b, key) in &self.pairs { + let (Some(ha), Some(hb)) = (&heads[*a], &heads[*b]) else { + continue; + }; + let eligible = node_ok[*a] && node_ok[*b]; + for id in &targets { + let (Some(x), Some(y)) = (ha.get(id), hb.get(id)) else { + continue; + }; + if x != y { + let entry = ( + (key.clone(), col.clone(), "M3", id.clone()), + json!({ "heads_a": x, "heads_b": y }), + ); + if eligible { + out.mismatches.push(entry.clone()); + } + out.all.push(entry); + } + } + } + // M5: plaintext parity on the same targets, only for encrypted profiles. + if !self.cfg.encrypted_fields.is_empty() { + let mut plain: Vec> = Vec::new(); + for (n, set) in ids.iter().enumerate() { + let Some(set) = set else { + plain.push(None); + continue; + }; + let mine: Vec = targets + .iter() + .filter(|t| set.contains(*t)) + .cloned() + .collect(); + let mut map = FieldMap::new(); + let mut failed = false; + for chunk in mine.chunks(self.cfg.batch) { + match self.encrypted_fields(n, &col, chunk).await { + Ok(part) => map.extend(part), + Err(_) => { + failed = true; + break; + } + } + } + if failed { + plain.push(None); + if !out.unreachable.contains(&n) { + out.unreachable.push(n); + } + node_ok[n] = false; + } else { + plain.push(Some(map)); + } + } + for (a, b, key) in &self.pairs { + let (Some(pa), Some(pb)) = (&plain[*a], &plain[*b]) else { + continue; + }; + let eligible = node_ok[*a] && node_ok[*b]; + out.m5_docs += pa.keys().filter(|id| pb.contains_key(*id)).count(); + for (id, detail) in m5_compare( + pa, + pb, + &self.nodes[*a].0, + &self.nodes[*b].0, + &targets, + &self.cfg.encrypted_fields, + ) { + let entry = ((key.clone(), col.clone(), "M5", id), detail); + if eligible { + out.mismatches.push(entry.clone()); + } + out.all.push(entry); + } + } + } + // M6: access parity on the protected subset, only for ACP profiles. + if self.cfg.acp.is_some() { + let protected: Vec = targets + .iter() + .filter(|id| self.doc_meta.get(*id).is_some_and(|m| m.1)) + .cloned() + .collect(); + let origin: HashMap = protected + .iter() + .map(|id| (id.clone(), self.doc_meta[id].0)) + .collect(); + let mut views: Vec> = Vec::new(); + for (n, set) in ids.iter().enumerate() { + let Some(set) = set else { + views.push(None); + continue; + }; + let mine: Vec = protected + .iter() + .filter(|t| set.contains(*t)) + .cloned() + .collect(); + let mut map = ViewMap::new(); + let mut failed = false; + for chunk in mine.chunks(self.cfg.batch) { + match self.views(n, &col, chunk).await { + Ok(part) => map.extend(part), + Err(_) => { + failed = true; + break; + } + } + } + if failed { + views.push(None); + if !out.unreachable.contains(&n) { + out.unreachable.push(n); + } + node_ok[n] = false; + } else { + views.push(Some(map)); + } + } + let names: Vec = self.nodes.iter().map(|(n, _)| n.clone()).collect(); + for (a, b, key) in &self.pairs { + let (Some(va), Some(vb)) = (&views[*a], &views[*b]) else { + continue; + }; + let eligible = node_ok[*a] && node_ok[*b]; + let (found, by_design) = + m6_compare(va, vb, &names, *a, *b, &origin, &protected); + out.m6_docs += va.keys().filter(|id| vb.contains_key(*id)).count() - by_design; + out.m6_by_design += by_design; + for (id, detail) in found { + let entry = ((key.clone(), col.clone(), "M6", id), detail); + if eligible { + out.mismatches.push(entry.clone()); + } + out.all.push(entry); + } + } + } + } + Ok(out) + } + + /// One record per (pair, collection, mechanism) confirmed in this check. + fn record(&mut self, confirmed: &[Confirmed], op_index: u64) -> Result<()> { + let mut groups: HashMap<(String, String, &'static str), Vec<&Confirmed>> = HashMap::new(); + for c in confirmed { + groups + .entry((c.0 .0.clone(), c.0 .1.clone(), c.0 .2)) + .or_default() + .push(c); + } + let mut keys: Vec<_> = groups.keys().cloned().collect(); + keys.sort(); + for (pair, col, mech) in keys { + let items = &groups[&(pair.clone(), col.clone(), mech)]; + let members: Vec<&str> = pair.split('|').collect(); + let doc_tags: Vec> = + items.iter().map(|i| self.tag_for(&pair, &i.0 .3)).collect(); + let mut tags: Vec = doc_tags.iter().flatten().copied().collect(); + tags.sort_by_key(|t| *t as u8); + tags.dedup(); + let untagged = doc_tags.iter().filter(|t| t.is_none()).count() as u64; + self.summary.tagged_docs += doc_tags.len() as u64 - untagged; + self.summary.untagged_docs += untagged; + let record = json!({ + "run_id": self.run_id, + "seed": self.seed, + "detected_wall_ts_ms": now_ms(), + "detected_op_index": op_index, + "pair": members, + "collection": col, + "mechanism": mech, + "doc_ids": items.iter().map(|i| &i.0 .3).collect::>(), + "details": items.iter().map(|i| &i.1).collect::>(), + "event_window": { + "op_index": [self.last_clear_op, op_index], + "topology_events": self.window_events, + }, + "tags": tags, + "doc_tags": doc_tags, + "confirmations": items.iter().map(|i| i.2).max().unwrap_or(0), + }); + serde_json::to_writer(&mut self.divergences, &record)?; + self.divergences.write_all(b"\n")?; + self.divergences.flush()?; + self.summary.divergences += 1; + println!( + "DIVERGENCE {mech} {col} on {pair}: {} doc(s) ({untagged} untagged), window ops {}..{}", + items.len(), + self.last_clear_op, + op_index + ); + } + Ok(()) + } + + /// Creates now visible on another node become lag samples for that + /// directed pair; the resolution is the check interval. + fn sample_lag(&mut self, ids: &[Option>]) -> Result<()> { + let now = now_ms(); + let op_index = self.op_index.load(Ordering::Relaxed); + let nodes = &self.nodes; + let total = nodes.len(); + let mut lines = Vec::new(); + self.awaiting.retain(|doc, (origin, created, seen)| { + for (n, set) in ids.iter().enumerate() { + if n == *origin || seen.contains(&n) { + continue; + } + if set.as_ref().is_some_and(|s| s.contains(doc)) { + seen.insert(n); + lines.push(json!({ + "wall_ts_ms": now, "op_index": op_index, "doc_id": doc, + "from": nodes[*origin].0, "to": nodes[n].0, + "lag_ms": now.saturating_sub(*created), "source": "poll", + })); + } + } + seen.len() < total - 1 && now.saturating_sub(*created) < LAG_TTL_MS + }); + for line in lines { + serde_json::to_writer(&mut self.lag, &line)?; + self.lag.write_all(b"\n")?; + } + self.lag.flush()?; + Ok(()) + } + + /// The owner bearer when ACP is configured (protected docs are invisible + /// to an anonymous read on their origin node), else none. + fn sweep_bearer(&mut self, url: &str) -> Result> { + match &mut self.tokens { + Some(t) => t.bearer(Actor::Owner, url), + None => Ok(None), + } + } + + async fn doc_ids(&mut self, node: usize, col: &str) -> Result> { + let (name, url) = self.nodes[node].clone(); + let bearer = self.sweep_bearer(&url)?; + let query = format!("{{ {col} {{ _docID }} }}"); + let data = gql_as(&self.http, &url, &query, bearer.as_deref()) + .await + .map_err(|e| eyre!("{name}: docID sweep of {col}: {e}"))?; + Ok(data[col] + .as_array() + .into_iter() + .flatten() + .filter_map(|d| d["_docID"].as_str().map(String::from)) + .collect()) + } + + /// Sorted head CIDs per docID, one POST for the whole chunk. + async fn heads(&mut self, node: usize, ids: &[String]) -> Result>> { + if ids.is_empty() { + return Ok(HashMap::new()); + } + let (name, url) = self.nodes[node].clone(); + let bearer = self.sweep_bearer(&url)?; + let selections: Vec = ids + .iter() + .enumerate() + .map(|(i, id)| format!("d{i}: _commits(docID: \"{id}\", depth: 1) {{ cid }}")) + .collect(); + let query = format!("{{ {} }}", selections.join(" ")); + let data = gql_as(&self.http, &url, &query, bearer.as_deref()) + .await + .map_err(|e| eyre!("{name}: head query for {} docs: {e}", ids.len()))?; + Ok(ids + .iter() + .enumerate() + .map(|(i, id)| { + let mut cids: Vec = data[format!("d{i}")] + .as_array() + .into_iter() + .flatten() + .filter_map(|c| c["cid"].as_str().map(String::from)) + .collect(); + cids.sort(); + (id.clone(), cids) + }) + .collect()) + } + + /// The configured encrypted fields for `ids`, one POST per chunk. + async fn encrypted_fields( + &mut self, + node: usize, + col: &str, + ids: &[String], + ) -> Result { + if ids.is_empty() { + return Ok(FieldMap::new()); + } + let (name, url) = self.nodes[node].clone(); + let bearer = self.sweep_bearer(&url)?; + let fields = self.cfg.encrypted_fields.join(" "); + let list: Vec = ids.iter().map(|id| format!("\"{id}\"")).collect(); + let query = format!( + "{{ {col}(filter: {{_docID: {{_in: [{}]}}}}) {{ _docID {fields} }} }}", + list.join(", ") + ); + let data = gql_as(&self.http, &url, &query, bearer.as_deref()) + .await + .map_err(|e| eyre!("{name}: encrypted-field read of {} docs: {e}", ids.len()))?; + Ok(data[col] + .as_array() + .into_iter() + .flatten() + .filter_map(|d| { + let id = d["_docID"].as_str()?.to_string(); + let vals = self + .cfg + .encrypted_fields + .iter() + .map(|f| d[f].as_str().map(String::from)) + .collect(); + Some((id, vals)) + }) + .collect()) + } + + /// Presence of `ids` as seen by owner, reader and anon: one POST per + /// viewer per chunk. A doc a viewer cannot see is simply absent. + async fn views(&mut self, node: usize, col: &str, ids: &[String]) -> Result { + if ids.is_empty() { + return Ok(ViewMap::new()); + } + let (name, url) = self.nodes[node].clone(); + let list: Vec = ids.iter().map(|id| format!("\"{id}\"")).collect(); + let query = format!( + "{{ {col}(filter: {{_docID: {{_in: [{}]}}}}) {{ _docID }} }}", + list.join(", ") + ); + let mut map: ViewMap = ids.iter().map(|id| (id.clone(), [false; 3])).collect(); + for (i, actor) in VIEWER_ACTORS.iter().enumerate() { + let bearer = match &mut self.tokens { + Some(t) => t.bearer(*actor, &url)?, + None => None, + }; + let data = gql_as(&self.http, &url, &query, bearer.as_deref()) + .await + .map_err(|e| eyre!("{name}: {} view of {} docs: {e}", VIEWERS[i], ids.len()))?; + for d in data[col].as_array().into_iter().flatten() { + if let Some(seen) = d["_docID"].as_str().and_then(|id| map.get_mut(id)) { + seen[i] = true; + } + } + } + Ok(map) + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn settle_holds_for_min_settle_then_exits_on_clear_or_budget() { + let (min, settle) = (Duration::from_secs(60), Duration::from_secs(120)); + // Clear early no longer ends the settle while the floor is unserved. + assert!(!settle_done(Duration::from_secs(13), min, settle, true)); + // Served floor plus clear ends it. + assert!(settle_done(Duration::from_secs(60), min, settle, true)); + // Never clear: the budget still ends it. + assert!(!settle_done(Duration::from_secs(119), min, settle, false)); + assert!(settle_done(Duration::from_secs(120), min, settle, false)); + // A floor past the budget wins: the budget alone cannot cut it short. + assert!(!settle_done( + Duration::from_secs(150), + Duration::from_secs(300), + settle, + true + )); + // Default: zero floor is the historical early exit on the first clear. + assert!(settle_done(Duration::ZERO, Duration::ZERO, settle, true)); + } + + fn fm(rows: &[(&str, &[Option<&str>])]) -> FieldMap { + rows.iter() + .map(|(id, vals)| { + ( + id.to_string(), + vals.iter().map(|v| v.map(String::from)).collect(), + ) + }) + .collect() + } + + #[test] + fn m5_equal_differ_undecryptable() { + let a = fm(&[ + ("d1", &[Some("s"), Some("1")]), + ("d2", &[Some("s"), Some("2")]), + ("d3", &[Some("s"), Some("3")]), + ]); + let b = fm(&[ + ("d1", &[Some("s"), Some("1")]), + ("d2", &[Some("x"), Some("2")]), + ("d3", &[None, Some("3")]), + ]); + let targets = [ + "d1".to_string(), + "d2".to_string(), + "d3".to_string(), + "d4".to_string(), + ]; + let fields = ["secret".to_string(), "pin".to_string()]; + let out = m5_compare(&a, &b, "rust-0", "go-0", &targets, &fields); + assert_eq!(out.len(), 2, "{out:?}"); + assert_eq!(out[0].0, "d2"); + assert_eq!(out[0].1["field"], "secret"); + assert!(out[0].1["a"].is_string() && out[0].1["b"].is_string()); + assert_eq!(out[1].0, "d3"); + assert_eq!( + out[1].1, + json!({ "undecryptable_on": "go-0", "field": "secret" }) + ); + } + + #[test] + fn m6_peer_pairs_compare_and_origin_pairs_are_by_design() { + let mut a = ViewMap::new(); + let mut b = ViewMap::new(); + a.insert("d1".into(), [true, true, false]); + b.insert("d1".into(), [true, true, false]); // equal + a.insert("d2".into(), [true, false, false]); + b.insert("d2".into(), [true, true, false]); // reader differs + a.insert("d3".into(), [true, false, false]); + b.insert("d3".into(), [true, true, true]); // origin vs peer + let mut origin = HashMap::new(); + origin.insert("d1".to_string(), 3usize); + origin.insert("d2".to_string(), 3); + origin.insert("d3".to_string(), 0); + let targets = [ + "d1".to_string(), + "d2".to_string(), + "d3".to_string(), + "d4".to_string(), + ]; + let names: Vec = ["rust-0", "rust-1", "go-0", "go-1"] + .iter() + .map(|s| s.to_string()) + .collect(); + let (out, by_design) = m6_compare(&a, &b, &names, 0, 2, &origin, &targets); + assert_eq!(by_design, 1, "d3: rust-0 is its origin, go-0 a peer"); + assert_eq!(out.len(), 1); + assert_eq!(out[0].0, "d2"); + assert_eq!( + out[0].1, + json!({ "viewer": "reader", "a": false, "b": true, "origin": "go-1" }) + ); + } +} diff --git a/crates/soak/src/churn.rs b/crates/soak/src/churn.rs new file mode 100644 index 0000000..9d3b749 --- /dev/null +++ b/crates/soak/src/churn.rs @@ -0,0 +1,504 @@ +//! Seeded topology churn (axis 2). +//! +//! The schedule of restart / crash-kill events is drawn at run start as +//! offsets on the run's churn clock, so it is a pure function of the seed +//! and printable before the run begins. The clock is wall time from workload +//! start, or virtual time (op progress, `op_index / rate`, the same clock the +//! generator's `virtual_ts_ms` uses) for manifests that predate the choice. +//! Down-time is always wall time so a stalled workload cannot leave a node +//! dead forever. + +use std::fs::File; +use std::io::{BufWriter, Write}; +use std::path::PathBuf; +use std::sync::atomic::{AtomicBool, AtomicU64, Ordering}; +use std::sync::Arc; +use std::time::{Duration, Instant}; + +use eyre::{Result, WrapErr}; +use rand::{rngs::StdRng, Rng, SeedableRng}; +use serde::{Deserialize, Serialize}; +use serde_json::{json, Value}; +use tokio::sync::{mpsc, oneshot}; + +use crate::executor::{gql, now_ms}; +use crate::meter::Meter; +use crate::nodes::Nodes; + +/// Stream derivation constant for the topology axis. +const TOPO_AXIS: u64 = 0x7090_10c4_0000_0002; + +#[derive(Clone, Copy, Debug, PartialEq, Eq, Serialize, Deserialize)] +#[serde(rename_all = "snake_case")] +pub enum ChurnKind { + /// SIGTERM, then start again on the same ports. + Restart, + /// SIGKILL, stay dead for `down_ms` of wall time, respawn. + CrashKill, + /// SIGTERM, stay stopped for `down_ms` with the ports held, start again. + GracefulLeave, + /// Network disconnect for `down_ms` of wall time, then connect. + Partition, +} + +/// Which clock the schedule offsets are read against. +#[derive(Clone, Copy, Debug, Default, PartialEq, Eq, Serialize, Deserialize)] +#[serde(rename_all = "snake_case")] +pub enum ChurnClock { + /// Op progress over the profile rate; the clock old manifests ran on. + #[default] + Virtual, + /// Elapsed wall time since the workload started. + Wall, +} + +#[derive(Clone, Debug, PartialEq, Serialize, Deserialize)] +pub struct ChurnEvent { + pub index: usize, + /// Offset from workload start on the run's churn clock. + pub virtual_ts_ms: u64, + pub node: usize, + pub kind: ChurnKind, + /// Zero for Restart. + pub down_ms: u64, +} + +#[derive(Clone, Debug, Serialize, Deserialize)] +pub struct ChurnConfig { + /// Mean time between events, mesh-wide. + pub spacing_ms: u64, + /// Minimum time from one event's end to the next on that node. + pub cooldown_ms: u64, + /// Down-time bounds, inclusive. + pub down_ms: (u64, u64), + /// Absent from old manifests, which ran on virtual time. + #[serde(default)] + pub clock: ChurnClock, +} + +impl Default for ChurnConfig { + fn default() -> Self { + Self { + spacing_ms: 120_000, + cooldown_ms: 120_000, + down_ms: (5_000, 30_000), + clock: ChurnClock::Virtual, + } + } +} + +/// Events in ascending clock time, all before `horizon_ms`. A draw that +/// lands on a cooling-down node is dropped, not redrawn, so the stream stays +/// a pure function of the seed. `partitions` adds the Partition kind to the +/// draw for backends that can cut a node off the network. +pub fn schedule_with( + seed: u64, + nodes: usize, + horizon_ms: u64, + cfg: &ChurnConfig, + partitions: bool, +) -> Vec { + let mut rng = StdRng::seed_from_u64(seed ^ TOPO_AXIS); + let mut last_end: Vec> = vec![None; nodes]; + let mut events = Vec::new(); + let mut t = 0; + loop { + t += rng.gen_range(cfg.spacing_ms / 2..=cfg.spacing_ms * 3 / 2); + if t >= horizon_ms { + return events; + } + let node = rng.gen_range(0..nodes); + let kinds = if partitions { 4 } else { 3 }; + let kind = match rng.gen_range(0..kinds) { + 0 => ChurnKind::Restart, + 1 => ChurnKind::CrashKill, + 2 => ChurnKind::GracefulLeave, + _ => ChurnKind::Partition, + }; + let down_ms = match kind { + ChurnKind::Restart => 0, + _ => rng.gen_range(cfg.down_ms.0..=cfg.down_ms.1), + }; + if last_end[node].is_some_and(|end| t < end + cfg.cooldown_ms) { + continue; + } + last_end[node] = Some(t + down_ms); + events.push(ChurnEvent { + index: events.len(), + virtual_ts_ms: t, + node, + kind, + down_ms, + }); + } +} + +/// A node went down or came back; the checker uses it for eligibility and +/// for the known-cause tags. +#[derive(Clone, Copy, Debug)] +pub struct Transition { + pub node: usize, + pub up: bool, + pub wall_ts_ms: u64, +} + +/// Virtual time of the workload: ops issued so far over the profile rate. +pub fn virtual_ms(op_index: u64, rate: f64) -> u64 { + (op_index as f64 * 1000.0 / rate) as u64 +} + +/// Fires `events` on `clock` until `workload_done`, then keeps metering +/// until `stop`, so the settle window is sampled too; returns with every node +/// up. Each down/up phase appends a line to the log. The meter samples from +/// here too, since this task holds the nodes. +#[allow(clippy::too_many_arguments)] +pub async fn run( + nodes: &mut Nodes, + events: Vec, + clock: ChurnClock, + rate: f64, + collection: String, + op_index: Arc, + transitions: mpsc::UnboundedSender, + log_path: PathBuf, + mut meter: Meter, + workload_done: Arc, + mut stop: oneshot::Receiver<()>, +) -> Result<()> { + let mut log = BufWriter::new( + File::create(&log_path).wrap_err_with(|| format!("creating {}", log_path.display()))?, + ); + let http = reqwest::Client::new(); + let started = Instant::now(); + let now = || match clock { + ChurnClock::Virtual => virtual_ms(op_index.load(Ordering::Relaxed), rate), + ChurnClock::Wall => started.elapsed().as_millis() as u64, + }; + let mut pending = events.into_iter().peekable(); + loop { + meter.maybe_sample(nodes).await?; + // Once the workload has stopped the run is settling: keep metering, + // but do not fire an event into the window the settle is measuring. + if !workload_done.load(Ordering::Relaxed) + && pending.peek().is_some_and(|e| e.virtual_ts_ms <= now()) + { + let event = pending.next().expect("peeked"); + fire( + nodes, + &event, + &http, + &collection, + &now, + &transitions, + &mut log, + ) + .await?; + continue; + } + tokio::select! { + _ = tokio::time::sleep(Duration::from_millis(250)) => {} + _ = &mut stop => { + meter.sample(nodes).await?; + return Ok(()); + } + } + } +} + +async fn fire( + nodes: &mut Nodes, + event: &ChurnEvent, + http: &reqwest::Client, + collection: &str, + now: &dyn Fn() -> u64, + transitions: &mpsc::UnboundedSender, + log: &mut BufWriter, +) -> Result<()> { + let name = nodes.name(event.node).to_string(); + let url = nodes.api_url(event.node); + let started = Instant::now(); + let _ = transitions.send(Transition { + node: event.node, + up: false, + wall_ts_ms: now_ms(), + }); + log_phase(log, event, &name, "down", now(), 0, None, None)?; + println!( + "churn #{} {:?} {name} at {}ms (planned {}ms)", + event.index, + event.kind, + now(), + event.virtual_ts_ms + ); + match event.kind { + ChurnKind::Restart => { + rotate_logs(nodes, event).await?; + nodes + .restart(event.node) + .await + .wrap_err_with(|| format!("{name}: restart"))? + } + ChurnKind::CrashKill => { + nodes + .kill(event.node) + .await + .wrap_err_with(|| format!("{name}: kill"))?; + tokio::time::sleep(Duration::from_millis(event.down_ms)).await; + nodes + .respawn(event.node) + .await + .wrap_err_with(|| format!("{name}: respawn"))?; + } + ChurnKind::GracefulLeave => { + rotate_logs(nodes, event).await?; + nodes + .stop(event.node) + .await + .wrap_err_with(|| format!("{name}: stop"))?; + tokio::time::sleep(Duration::from_millis(event.down_ms)).await; + nodes + .start_stopped(event.node) + .await + .wrap_err_with(|| format!("{name}: start after leave"))?; + } + ChurnKind::Partition => { + nodes + .partition(event.node) + .await + .wrap_err_with(|| format!("{name}: partition"))?; + tokio::time::sleep(Duration::from_millis(event.down_ms)).await; + nodes + .rejoin(event.node) + .await + .wrap_err_with(|| format!("{name}: rejoin"))?; + log_phase( + log, + event, + &name, + "rejoin", + now(), + started.elapsed().as_millis() as u64, + None, + Some(&nodes.p2p_addr(event.node)), + )?; + } + } + wait_healthy(http, &url, collection, Duration::from_secs(60)) + .await + .wrap_err_with(|| format!("{name}: not healthy after {:?}", event.kind))?; + let _ = transitions.send(Transition { + node: event.node, + up: true, + wall_ts_ms: now_ms(), + }); + let pid = peer_id(http, &url).await; + log_phase( + log, + event, + &name, + "up", + now(), + started.elapsed().as_millis() as u64, + pid.as_deref(), + None, + )?; + println!( + "churn #{} {name} back up after {:?} as peer {}", + event.index, + started.elapsed(), + pid.as_deref().unwrap_or("?") + ); + Ok(()) +} + +/// The node's libp2p peer ID from `GET /api/v0/p2p/info`. Both runtimes +/// answer with a list of multiaddrs; take the id after the first `/p2p/`. +pub async fn peer_id(http: &reqwest::Client, url: &str) -> Option { + let body: Value = http + .get(format!("{url}/api/v0/p2p/info")) + .send() + .await + .ok()? + .json() + .await + .ok()?; + find_peer_id(&body) +} + +fn find_peer_id(v: &Value) -> Option { + match v { + Value::String(s) => s + .split("/p2p/") + .nth(1) + .map(|id| id.split('/').next().unwrap_or(id).to_string()), + Value::Array(a) => a.iter().find_map(find_peer_id), + Value::Object(o) => o.values().find_map(find_peer_id), + _ => None, + } +} + +/// The harness truncates stdout.log on every spawn; keep the old process's +/// log so the artifact holds the whole history. A container's output is +/// flushed to the same files first. +async fn rotate_logs(nodes: &mut Nodes, event: &ChurnEvent) -> Result<()> { + nodes.dump_logs(event.node).await?; + let name = nodes.name(event.node).to_string(); + let log_dir = nodes.log_dir(event.node); + for file in ["stdout.log", "stderr.log"] { + let from = log_dir.join(file); + let to = log_dir.join(format!("{file}.before-event-{}", event.index)); + if from.exists() { + std::fs::rename(&from, &to) + .wrap_err_with(|| format!("{name}: rotating {}", from.display()))?; + } + } + Ok(()) +} + +/// `p2p_addr` is the container's re-read address, recorded on the `rejoin` +/// phase only. +#[allow(clippy::too_many_arguments)] +fn log_phase( + log: &mut BufWriter, + event: &ChurnEvent, + node: &str, + phase: &str, + virtual_ts_ms: u64, + duration_ms: u64, + peer_id: Option<&str>, + p2p_addr: Option<&str>, +) -> Result<()> { + let mut line = json!({ + "event": event.index, "kind": event.kind, "node": node, "phase": phase, + "planned_virtual_ts_ms": event.virtual_ts_ms, "virtual_ts_ms": virtual_ts_ms, + "wall_ts_ms": now_ms(), "down_ms": event.down_ms, "duration_ms": duration_ms, + "peer_id": peer_id, + }); + if let Some(addr) = p2p_addr { + line["p2p_addr"] = json!(addr); + } + serde_json::to_writer(&mut *log, &line)?; + log.write_all(b"\n")?; + log.flush()?; + Ok(()) +} + +async fn wait_healthy( + http: &reqwest::Client, + url: &str, + collection: &str, + timeout: Duration, +) -> Result<()> { + let query = format!("{{ {collection}(limit: 1) {{ _docID }} }}"); + let deadline = Instant::now() + timeout; + loop { + if gql(http, url, &query).await.is_ok() { + return Ok(()); + } + eyre::ensure!( + Instant::now() < deadline, + "no healthy GraphQL answer within {timeout:?}" + ); + tokio::time::sleep(Duration::from_millis(250)).await; + } +} + +#[cfg(test)] +mod tests { + use super::*; + + const HOUR: u64 = 3_600_000; + + fn plan(seed: u64) -> Vec { + schedule_with(seed, 2, HOUR, &ChurnConfig::default(), false) + } + + #[test] + fn same_seed_same_schedule() { + assert_eq!(plan(5), plan(5)); + assert!( + plan(5).len() >= 10, + "an hour at ~2min spacing yields many events" + ); + } + + #[test] + fn different_seed_different_schedule() { + assert_ne!(plan(5), plan(6)); + } + + #[test] + fn ordered_within_horizon_and_cooldown() { + let cfg = ChurnConfig::default(); + let events = plan(9); + let mut last_end = [0u64; 2]; + let mut prev = 0; + for e in &events { + assert!(e.virtual_ts_ms < HOUR); + assert!(e.virtual_ts_ms >= prev, "events must be time-ordered"); + assert!( + e.virtual_ts_ms >= last_end[e.node] + cfg.cooldown_ms || last_end[e.node] == 0, + "node {} event at {} violates cooldown after {}", + e.node, + e.virtual_ts_ms, + last_end[e.node] + ); + last_end[e.node] = e.virtual_ts_ms + e.down_ms; + prev = e.virtual_ts_ms; + } + } + + #[test] + fn down_time_within_bounds_and_all_kinds_drawn() { + let cfg = ChurnConfig::default(); + let events = plan(3); + for kind in [ + ChurnKind::Restart, + ChurnKind::CrashKill, + ChurnKind::GracefulLeave, + ] { + assert!( + events.iter().any(|e| e.kind == kind), + "{kind:?} never drawn" + ); + } + for e in &events { + match e.kind { + ChurnKind::Restart => assert_eq!(e.down_ms, 0), + _ => assert!((cfg.down_ms.0..=cfg.down_ms.1).contains(&e.down_ms)), + } + } + } + + /// Replay redraws the schedule from the seed, so the draws without the + /// Partition kind are frozen: these are the first events of seed 106. + #[test] + fn process_schedule_is_unchanged_without_partition() { + let ev = schedule_with(106, 4, 1_800_000, &ChurnConfig::default(), false); + let head: Vec<_> = ev + .iter() + .take(4) + .map(|e| (e.virtual_ts_ms, e.node, e.kind, e.down_ms)) + .collect(); + assert_eq!( + head, + [ + (174_829, 3, ChurnKind::CrashKill, 26_658), + (348_896, 2, ChurnKind::CrashKill, 15_703), + (470_098, 3, ChurnKind::GracefulLeave, 14_878), + (635_362, 2, ChurnKind::Restart, 0), + ] + ); + } + + #[test] + fn docker_schedule_draws_partitions() { + let cfg = ChurnConfig::default(); + let ev = schedule_with(106, 6, 1_800_000, &cfg, true); + assert!(ev.iter().any(|e| e.kind == ChurnKind::Partition)); + assert!(ev + .iter() + .filter(|e| e.kind == ChurnKind::Partition) + .all(|e| (cfg.down_ms.0..=cfg.down_ms.1).contains(&e.down_ms))); + } +} diff --git a/crates/soak/src/confirm.rs b/crates/soak/src/confirm.rs new file mode 100644 index 0000000..8640d78 --- /dev/null +++ b/crates/soak/src/confirm.rs @@ -0,0 +1,199 @@ +//! Mismatch confirmation: a mismatch becomes a divergence only after it has +//! been seen in `confirmations` consecutive checks spanning at least `grace`. +//! This is what separates "sync in flight" from "diverged" without an event +//! bus on the Go side. +use std::collections::{HashMap, HashSet}; +use std::time::{Duration, Instant}; + +/// (pair, collection, mechanism, docID) identifies one mismatch; the pair is +/// `"|"` with a < b. +pub type Key = (String, String, &'static str, String); + +struct Pending { + first_seen: Instant, + count: u32, + emitted: bool, +} + +pub struct Confirmer { + confirmations: u32, + grace: Duration, + pending: HashMap, +} + +impl Confirmer { + pub fn new(confirmations: u32, grace: Duration) -> Self { + Self { + confirmations, + grace, + pending: HashMap::new(), + } + } + + /// Feed one check's mismatches. Returns the keys that just became + /// confirmed divergences, with their latest detail and check count. + /// A key absent from `mismatches` is cleared, unless `frozen` says its + /// pair was not eligible this check (a member down or in grace): frozen + /// keys are neither counted nor cleared, and any of them present in + /// `mismatches` are ignored. + pub fn observe( + &mut self, + now: Instant, + mismatches: Vec<(Key, D)>, + frozen: impl Fn(&Key) -> bool, + ) -> Vec<(Key, D, u32)> { + let seen: HashSet<&Key> = mismatches.iter().map(|(k, _)| k).collect(); + self.pending.retain(|k, _| seen.contains(k) || frozen(k)); + let mut confirmed = Vec::new(); + for (key, detail) in mismatches { + if frozen(&key) { + continue; + } + let p = self.pending.entry(key.clone()).or_insert(Pending { + first_seen: now, + count: 0, + emitted: false, + }); + p.count += 1; + if !p.emitted + && p.count >= self.confirmations + && now.duration_since(p.first_seen) >= self.grace + { + p.emitted = true; + confirmed.push((key, detail, p.count)); + } + } + confirmed + } + + /// Mismatches seen but not (yet) confirmed. + pub fn pending(&self) -> usize { + self.pending.values().filter(|p| !p.emitted).count() + } + + /// Confirmed divergences whose mismatch is still present. + pub fn unresolved(&self) -> usize { + self.pending.values().filter(|p| p.emitted).count() + } +} + +#[cfg(test)] +mod tests { + use super::*; + + fn key(id: &str) -> Key { + ("a|b".to_string(), "Users".to_string(), "M3", id.to_string()) + } + + fn never(_: &Key) -> bool { + false + } + + fn secs(n: u64) -> Duration { + Duration::from_secs(n) + } + + #[test] + fn confirms_after_r_consecutive_checks() { + let mut c = Confirmer::new(3, secs(0)); + let t0 = Instant::now(); + assert!(c.observe(t0, vec![(key("a"), 1)], never).is_empty()); + assert!(c + .observe(t0 + secs(1), vec![(key("a"), 2)], never) + .is_empty()); + let out = c.observe(t0 + secs(2), vec![(key("a"), 3)], never); + assert_eq!(out, vec![(key("a"), 3, 3)]); + assert!( + c.observe(t0 + secs(3), vec![(key("a"), 4)], never) + .is_empty(), + "no re-emit" + ); + } + + #[test] + fn grace_window_delays_confirmation() { + let mut c = Confirmer::new(1, secs(10)); + let t0 = Instant::now(); + assert!(c.observe(t0, vec![(key("a"), ())], never).is_empty()); + assert!(c + .observe(t0 + secs(5), vec![(key("a"), ())], never) + .is_empty()); + assert_eq!( + c.observe(t0 + secs(10), vec![(key("a"), ())], never).len(), + 1 + ); + } + + #[test] + fn a_clear_check_resets_the_count() { + let mut c = Confirmer::new(2, secs(0)); + let t0 = Instant::now(); + assert!(c.observe(t0, vec![(key("a"), ())], never).is_empty()); + assert!(c.observe::<()>(t0 + secs(1), Vec::new(), never).is_empty()); + assert_eq!(c.pending(), 0); + assert!(c + .observe(t0 + secs(2), vec![(key("a"), ())], never) + .is_empty()); + assert_eq!( + c.observe(t0 + secs(3), vec![(key("a"), ())], never).len(), + 1 + ); + } + + #[test] + fn unresolved_counts_emitted_mismatches_still_present() { + let mut c = Confirmer::new(1, secs(0)); + let t0 = Instant::now(); + assert_eq!(c.observe(t0, vec![(key("a"), ())], never).len(), 1); + assert_eq!(c.unresolved(), 1); + c.observe::<()>(t0 + secs(1), Vec::new(), never); + assert_eq!(c.unresolved(), 0); + } + + /// A frozen key (its pair ineligible this check) keeps its count when + /// absent and gains nothing when present; other keys proceed normally. + #[test] + fn frozen_keys_are_neither_cleared_nor_counted() { + let mut c = Confirmer::new(2, secs(0)); + let t0 = Instant::now(); + let frozen = |k: &Key| k.0 == "a|b"; + let other = ( + "a|c".to_string(), + "Users".to_string(), + "M3", + "x".to_string(), + ); + assert!(c + .observe(t0, vec![(key("a"), ()), (other.clone(), ())], never) + .is_empty()); + // Check 2: pair a|b frozen. "a" absent must not clear; "a" present must not count. + assert!( + c.observe(t0 + secs(1), vec![(other.clone(), ())], frozen) + .len() + == 1 + ); + assert!(c + .observe(t0 + secs(2), vec![(key("a"), ())], frozen) + .is_empty()); + // Check 4: eligible again; "a" now has count 2 -> confirmed. + assert_eq!( + c.observe(t0 + secs(3), vec![(key("a"), ())], never).len(), + 1 + ); + } + + #[test] + fn heal_then_recur_emits_again() { + let mut c = Confirmer::new(1, secs(0)); + let t0 = Instant::now(); + assert_eq!(c.observe(t0, vec![(key("a"), ())], never).len(), 1); + assert!(c + .observe(t0 + secs(1), vec![(key("a"), ())], never) + .is_empty()); + assert!(c.observe::<()>(t0 + secs(2), Vec::new(), never).is_empty()); + assert_eq!( + c.observe(t0 + secs(3), vec![(key("a"), ())], never).len(), + 1 + ); + } +} diff --git a/crates/soak/src/executor.rs b/crates/soak/src/executor.rs new file mode 100644 index 0000000..6cfc65a --- /dev/null +++ b/crates/soak/src/executor.rs @@ -0,0 +1,637 @@ +//! Executes planned ops over HTTP GraphQL and appends one JSONL record each. + +use std::fs::File; +use std::io::{BufWriter, Write}; +use std::path::{Path, PathBuf}; +use std::time::{Duration, Instant, SystemTime, UNIX_EPOCH}; + +use eyre::{Result, WrapErr}; +use serde::Serialize; +use serde_json::{json, Value}; + +use crate::auth::{host_port, Identities, TokenCache}; +use crate::generator::{Actor, OpKind, PlannedOp, Profile}; + +/// One executed op. `wall_ts_ms`, `latency_ms` and `error` (runtime text, +/// e.g. Go's "did you mean" list is unordered) are the only fields a replay +/// may differ in. +#[derive(Debug, Serialize)] +pub struct OpRecord { + pub op_index: u64, + pub virtual_ts_ms: u64, + pub wall_ts_ms: u64, + pub node: String, + pub kind: OpKind, + pub collection: String, + pub doc_id: Option, + pub ok: bool, + /// The victim's create failed (its node was down), so there was nothing + /// to update or delete; not an error of this op. + pub skipped: bool, + pub error: Option, + pub latency_ms: u64, + /// Identity the request was sent as (`None` = anonymous / pre-ACP record). + pub actor: Option, + /// `"http"` (GraphQL) or `"cli"` (grants go through the node binary). + pub path: &'static str, + /// Bytes of the GraphQL input this op sent; 0 for ops that send none + /// (queries, grants, deletes). + /// `#[serde(default)]` is a no-op today -- `OpRecord` only derives `Serialize` -- kept in case it gains `Deserialize`. + #[serde(default)] + pub payload_bytes: usize, +} + +pub struct Executor { + http: reqwest::Client, + /// (name, api_url) per node index. + nodes: Vec<(String, String)>, + collection: String, + encrypt_fields: Vec, + se_field: Option, + /// Slot -> docID learned from the create response. + slots: Vec>, + /// Bearer tokens per actor; `None` when the profile has no identities. + tokens: Option, + /// Node binary per node index, for CLI-only ops (grants). + binaries: Vec, + acp: bool, + log: BufWriter, +} + +impl Executor { + pub fn new( + nodes: Vec<(String, String)>, + profile: &Profile, + identities: Option, + binaries: Vec, + log_path: &Path, + ) -> Result { + let log = + File::create(log_path).wrap_err_with(|| format!("creating {}", log_path.display()))?; + Ok(Self { + http: http_client(Duration::from_secs(30)), + nodes, + collection: profile.collection.clone(), + encrypt_fields: profile.encrypt_fields.clone(), + se_field: profile.se_field.clone(), + slots: Vec::new(), + tokens: identities.map(TokenCache::new), + binaries, + acp: profile.is_acp(), + log: BufWriter::new(log), + }) + } + + /// Run one op against its node and log the record. `Err` only for log I/O. + pub async fn execute(&mut self, op: &PlannedOp) -> Result { + if self.acp && op.kind == OpKind::Query { + return self.viewer_query(op).await; + } + if op.kind == OpKind::Grant { + return self.grant(op).await; + } + let (name, url) = &self.nodes[op.node]; + let col = op + .collection + .as_deref() + .unwrap_or(self.collection.as_str()) + .to_string(); + let mut payload = op.payload.clone().unwrap_or_else(|| "{}".into()); + let parent_orphan = match op.parent_slot { + Some(ps) => match self.slots.get(ps).cloned().flatten() { + Some(id) => { + payload = with_author(&payload, &id); + false + } + None => true, + }, + None => false, + }; + let victim = op.slot.and_then(|s| self.slots.get(s).cloned().flatten()); + let expected: Vec = op + .expect_slots + .iter() + .filter_map(|s| self.slots.get(*s).cloned().flatten()) + .collect(); + let query = if parent_orphan { + None + } else { + match op.kind { + OpKind::Create => Some(create_mutation(&col, &payload, &self.encrypt_fields)), + OpKind::Update => victim.as_ref().map(|id| { + format!( + "mutation {{ update_{col}(docID: \"{id}\", input: {payload}) {{ _docID }} }}" + ) + }), + OpKind::Delete => victim + .as_ref() + .map(|id| format!("mutation {{ delete_{col}(docID: \"{id}\") {{ _docID }} }}")), + OpKind::Query => Some(match &self.se_field { + Some(field) => se_query(&col, field, &payload), + None => format!("{{ {payload} }}"), + }), + OpKind::Grant => unreachable!("grants return early"), + } + }; + let bearer = match (op.actor, self.tokens.as_mut()) { + (Some(a), Some(t)) => t.bearer(a, url).map_err(|e| e.to_string()), + _ => Ok(None), + }; + + let wall_ts_ms = now_ms(); + let started = Instant::now(); + let skipped = query.is_none(); + let outcome = match (query, bearer) { + (Some(q), Ok(b)) => gql_as(&self.http, url, &q, b.as_deref()).await, + (Some(_), Err(e)) => Err(e), + (None, _) => Err("orphan: this slot's create failed".to_string()), + }; + let latency_ms = started.elapsed().as_millis() as u64; + + let mut doc_id = victim; + let (ok, error) = match outcome { + Ok(data) => match op.kind { + OpKind::Create => { + // Both runtimes answer `add_X` with a list; tolerate an object. + let added = &data[format!("add_{col}")]; + let first = added.as_array().and_then(|a| a.first()).unwrap_or(added); + doc_id = first["_docID"].as_str().map(String::from); + match (&doc_id, op.slot) { + (Some(id), Some(slot)) => { + if self.slots.len() <= slot { + self.slots.resize(slot + 1, None); + } + self.slots[slot] = Some(id.clone()); + (true, None) + } + _ => (false, Some("create returned no _docID".to_string())), + } + } + // A docID the node does not hold yet (replication lag) is + // not an error to either runtime: the reply is just empty. + OpKind::Update | OpKind::Delete => { + let verb = if op.kind == OpKind::Update { + "update" + } else { + "delete" + }; + let matched = match &data[format!("{verb}_{col}")] { + Value::Array(a) => a.len(), + Value::Object(_) => 1, + _ => 0, + }; + if matched > 0 { + (true, None) + } else { + ( + false, + Some("no doc matched on this node (not replicated yet?)".to_string()), + ) + } + } + OpKind::Query => match &self.se_field { + Some(_) => se_verdict(&data, &col, &expected), + None => (true, None), + }, + OpKind::Grant => unreachable!("grants return early"), + }, + Err(e) => (false, Some(e)), + }; + + let record = OpRecord { + op_index: op.index, + virtual_ts_ms: op.virtual_ts_ms, + wall_ts_ms, + node: name.clone(), + kind: op.kind, + collection: col.clone(), + doc_id, + ok, + skipped, + error, + latency_ms, + actor: op.actor, + path: "http", + // Create/update send `payload` as the mutation's input object; + // delete and plain query send no document payload. + payload_bytes: match op.kind { + OpKind::Create | OpKind::Update => payload.len(), + _ => 0, + }, + }; + self.write(&record)?; + Ok(record) + } + + /// Read the victim as owner, reader and anonymous on `op.node`; ok iff every + /// read answered (the visibility itself is the checker's judgement). On + /// success `error` carries the observed views so `ops.jsonl` records them. + async fn viewer_query(&mut self, op: &PlannedOp) -> Result { + let (name, url) = self.nodes[op.node].clone(); + let wall_ts_ms = now_ms(); + let started = Instant::now(); + let victim = op.slot.and_then(|s| self.slots.get(s).cloned().flatten()); + let (ok, skipped, error) = match (&victim, self.tokens.as_mut()) { + (None, _) => ( + false, + true, + Some("orphan: this slot's create failed".to_string()), + ), + (_, None) => ( + false, + false, + Some("viewer query without identities".to_string()), + ), + (Some(id), Some(tokens)) => { + let q = format!( + "{{ {}(filter: {{_docID: {{_eq: \"{id}\"}}}}) {{ _docID }} }}", + self.collection + ); + let mut seen = Vec::new(); + let mut err = None; + for actor in [Actor::Owner, Actor::Reader, Actor::Anon] { + match tokens.bearer(actor, &url).map_err(|e| e.to_string()) { + Err(e) => { + err = Some(e); + break; + } + Ok(b) => match gql_as(&self.http, &url, &q, b.as_deref()).await { + Ok(data) => { + seen.push(data[&self.collection].as_array().map_or(0, Vec::len) > 0) + } + Err(e) => { + err = Some(format!("{actor:?}: {e}")); + break; + } + }, + } + } + match err { + Some(e) => (false, false, Some(e)), + None => (true, false, Some(views_line(&seen))), + } + } + }; + let record = OpRecord { + op_index: op.index, + virtual_ts_ms: op.virtual_ts_ms, + wall_ts_ms, + node: name, + kind: op.kind, + collection: self.collection.clone(), + doc_id: victim, + ok, + skipped, + error, + latency_ms: started.elapsed().as_millis() as u64, + actor: None, + path: "http", + // Viewer query reads the victim by docID; it sends no document payload. + payload_bytes: 0, + }; + self.write(&record)?; + Ok(record) + } + + /// `acp document relationship add` through the node's CLI (no HTTP route in the harness). + async fn grant(&mut self, op: &PlannedOp) -> Result { + let (name, url) = self.nodes[op.node].clone(); + let wall_ts_ms = now_ms(); + let started = Instant::now(); + let victim = op.slot.and_then(|s| self.slots.get(s).cloned().flatten()); + let ids = self.tokens.as_ref().map(|t| t.identities().clone()); + let (ok, skipped, error) = match (&victim, ids) { + (None, _) => ( + false, + true, + Some("orphan: this slot's create failed".to_string()), + ), + (_, None) => (false, false, Some("grant without identities".to_string())), + (Some(id), Some(ids)) => { + let out = tokio::process::Command::new(&self.binaries[op.node]) + .args([ + "--url", + host_port(&url), + "client", + "-i", + &ids.owner.key_hex, + "acp", + "document", + "relationship", + "add", + "-c", + &self.collection, + "--docID", + id, + "-r", + "reader", + "-a", + &ids.reader.did, + ]) + .output() + .await; + match out { + Ok(o) if o.status.success() => (true, false, None), + Ok(o) => ( + false, + false, + Some(format!( + "cli: {}", + String::from_utf8_lossy(&o.stderr).trim() + )), + ), + Err(e) => (false, false, Some(format!("cli spawn: {e}"))), + } + } + }; + let record = OpRecord { + op_index: op.index, + virtual_ts_ms: op.virtual_ts_ms, + wall_ts_ms, + node: name, + kind: op.kind, + collection: self.collection.clone(), + doc_id: victim, + ok, + skipped, + error, + latency_ms: started.elapsed().as_millis() as u64, + actor: Some(Actor::Owner), + path: "cli", + // Grant sends a relationship add via the CLI, not a document payload. + payload_bytes: 0, + }; + self.write(&record)?; + Ok(record) + } + + fn write(&mut self, record: &OpRecord) -> Result<()> { + serde_json::to_writer(&mut self.log, record)?; + self.log.write_all(b"\n")?; + self.log.flush()?; + Ok(()) + } +} + +/// The `error` text of a successful viewer query: `[owner, reader, anon]` saw the doc. +pub fn views_line(seen: &[bool]) -> String { + format!( + "views owner={} reader={} anon={}", + seen[0], seen[1], seen[2] + ) +} + +/// Splice the soak relation spelling into a Book create payload. +pub fn with_author(payload: &str, id: &str) -> String { + let inner = payload + .trim() + .strip_prefix('{') + .and_then(|s| s.strip_suffix('}')) + .unwrap_or(payload) + .trim(); + if inner.is_empty() { + format!("{{author: \"{id}\"}}") + } else { + format!("{{{inner}, author: \"{id}\"}}") + } +} + +/// `add_` with the profile's `encryptFields:` list (unquoted names). +pub fn create_mutation(col: &str, payload: &str, encrypt_fields: &[String]) -> String { + if encrypt_fields.is_empty() { + format!("mutation {{ add_{col}(input: [{payload}]) {{ _docID }} }}") + } else { + format!( + "mutation {{ add_{col}(input: [{payload}], encryptFields: [{}]) {{ _docID }} }}", + encrypt_fields.join(", ") + ) + } +} + +/// Searchable-encryption equality query; the owner fans it to its replicators. +pub fn se_query(col: &str, field: &str, value: &str) -> String { + format!("{{ encrypted_{col}(filter: {{{field}: {{_eq: \"{value}\"}}}}) {{ docIDs }} }}") +} + +/// ok iff every expected docID is in the flattened `docIDs` of the reply. +pub fn se_verdict(data: &Value, col: &str, expected: &[String]) -> (bool, Option) { + let got: std::collections::HashSet<&str> = data[format!("encrypted_{col}")] + .as_array() + .into_iter() + .flatten() + .flat_map(|row| row["docIDs"].as_array().into_iter().flatten()) + .filter_map(Value::as_str) + .collect(); + let missing: Vec<&str> = expected + .iter() + .map(String::as_str) + .filter(|id| !got.contains(id)) + .collect(); + if missing.is_empty() { + return (true, None); + } + let mut shown: Vec<&str> = missing.iter().copied().take(5).collect(); + if missing.len() > 5 { + shown.push("..."); + } + ( + false, + Some(format!( + "se query missing {} of {} expected docIDs: {}", + missing.len(), + expected.len(), + shown.join(", ") + )), + ) +} + +pub fn now_ms() -> u64 { + SystemTime::now() + .duration_since(UNIX_EPOCH) + .map_or(0, |d| d.as_millis() as u64) +} + +/// A client whose requests fail after `timeout`; without one a request that +/// never completes stalls the workload loop for the rest of the run. +pub(crate) fn http_client(timeout: Duration) -> reqwest::Client { + reqwest::Client::builder() + .timeout(timeout) + .build() + .expect("reqwest client") +} + +/// POST a GraphQL document to a node; transport and GraphQL errors become +/// a message so the caller can record them instead of failing the run. +pub async fn gql(http: &reqwest::Client, url: &str, query: &str) -> Result { + gql_as(http, url, query, None).await +} + +/// `gql` with an optional bearer token (identity-scoped request). +pub async fn gql_as( + http: &reqwest::Client, + url: &str, + query: &str, + bearer: Option<&str>, +) -> Result { + let mut req = http + .post(format!("{url}/api/v0/graphql")) + .json(&json!({ "query": query })); + if let Some(b) = bearer { + req = req.bearer_auth(b); + } + let resp = req.send().await.map_err(|e| format!("http: {e}"))?; + let status = resp.status(); + let text = resp.text().await.map_err(|e| format!("http body: {e}"))?; + if !status.is_success() { + let snippet: String = text.chars().take(200).collect(); + return Err(format!("http {status}: {snippet}")); + } + let body: Value = serde_json::from_str(&text).map_err(|e| format!("bad json: {e}"))?; + if let Some(errors) = body + .get("errors") + .and_then(Value::as_array) + .filter(|e| !e.is_empty()) + { + let msgs: Vec<&str> = errors + .iter() + .map(|e| e["message"].as_str().unwrap_or("?")) + .collect(); + return Err(format!("graphql: {}", msgs.join("; "))); + } + Ok(body["data"].clone()) +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn gql_times_out_against_silent_listener() { + let listener = std::net::TcpListener::bind("127.0.0.1:0").unwrap(); + let url = format!("http://{}", listener.local_addr().unwrap()); + let http = http_client(Duration::from_secs(1)); + let started = Instant::now(); + let out = + tokio::runtime::Runtime::new() + .unwrap() + .block_on(gql(&http, &url, "{ __typename }")); + let err = out.expect_err("a request nobody answers must fail"); + assert!(err.starts_with("http:"), "{err}"); + assert!(started.elapsed() < Duration::from_secs(5)); + } + + #[test] + fn gql_as_reports_non_success_status() { + let listener = std::net::TcpListener::bind("127.0.0.1:0").unwrap(); + let url = format!("http://{}", listener.local_addr().unwrap()); + std::thread::spawn(move || { + use std::io::{Read, Write}; + let (mut sock, _) = listener.accept().unwrap(); + let mut buf = [0u8; 4096]; + let _ = sock.read(&mut buf); + let body = r#"{"error":"nope"}"#; + let resp = format!( + "HTTP/1.1 403 Forbidden\r\nContent-Type: application/json\r\nContent-Length: {}\r\nConnection: close\r\n\r\n{body}", + body.len() + ); + sock.write_all(resp.as_bytes()).unwrap(); + }); + let http = http_client(Duration::from_secs(5)); + let out = tokio::runtime::Runtime::new().unwrap().block_on(gql_as( + &http, + &url, + "{ __typename }", + Some("bad"), + )); + let err = out.expect_err("a 403 must not parse as Ok"); + assert!(err.starts_with("http 403"), "{err}"); + assert!(err.contains("nope"), "{err}"); + } + + #[test] + fn op_record_carries_payload_bytes() { + let json = serde_json::to_value(OpRecord { + op_index: 0, + virtual_ts_ms: 0, + wall_ts_ms: 0, + node: "rust-0".into(), + kind: OpKind::Create, + collection: "Users".into(), + doc_id: None, + ok: true, + skipped: false, + error: None, + latency_ms: 0, + actor: None, + path: "http", + payload_bytes: 1234, + }) + .unwrap(); + assert_eq!(json["payload_bytes"], 1234); + } + + #[test] + fn views_line_format() { + assert_eq!( + views_line(&[true, false, false]), + "views owner=true reader=false anon=false" + ); + } + + #[test] + fn with_author_uses_the_soak_spelling() { + assert_eq!( + with_author("{name: \"x\", rating: 1.0}", "bae-parent"), + "{name: \"x\", rating: 1.0, author: \"bae-parent\"}" + ); + } + + #[test] + fn create_mutation_plain_and_encrypted() { + assert_eq!( + create_mutation("Users", "{a: 1}", &[]), + "mutation { add_Users(input: [{a: 1}]) { _docID } }" + ); + assert_eq!( + create_mutation( + "Vault", + "{a: 1}", + &["secret".to_string(), "pin".to_string()] + ), + "mutation { add_Vault(input: [{a: 1}], encryptFields: [secret, pin]) { _docID } }" + ); + } + + #[test] + fn se_query_string() { + assert_eq!( + se_query("Vault", "name", "name-07"), + "{ encrypted_Vault(filter: {name: {_eq: \"name-07\"}}) { docIDs } }" + ); + } + + #[test] + fn se_verdict_subset_ok_missing_fails() { + let data = json!({ "encrypted_Vault": [ { "docIDs": ["bae-a", "bae-b"] }, { "docIDs": ["bae-c"] } ] }); + let want = ["bae-a".to_string(), "bae-c".to_string()]; + assert_eq!(se_verdict(&data, "Vault", &want), (true, None)); + let want2 = ["bae-a".to_string(), "bae-z".to_string()]; + let (ok, err) = se_verdict(&data, "Vault", &want2); + assert!(!ok); + assert_eq!( + err.as_deref(), + Some("se query missing 1 of 2 expected docIDs: bae-z") + ); + assert_eq!( + se_verdict(&json!({ "encrypted_Vault": [] }), "Vault", &[]), + (true, None) + ); + let many: Vec = (0..7).map(|i| format!("bae-m{i}")).collect(); + let (ok, err) = se_verdict(&json!({ "encrypted_Vault": [] }), "Vault", &many); + assert!(!ok); + assert_eq!( + err.as_deref(), + Some("se query missing 7 of 7 expected docIDs: bae-m0, bae-m1, bae-m2, bae-m3, bae-m4, ...") + ); + } +} diff --git a/crates/soak/src/generator.rs b/crates/soak/src/generator.rs new file mode 100644 index 0000000..b516008 --- /dev/null +++ b/crates/soak/src/generator.rs @@ -0,0 +1,1285 @@ +//! Seeded data-op planner (axis 1). +//! +//! Pure: the op sequence is a function of (seed, profile, node count) only. +//! Execution outcomes never feed back, so a replay from the same seed plans +//! the identical sequence even if some ops failed the first time. Victims for +//! update/delete are ledger *slots* (creation order); the executor maps slots +//! to the docIDs it learned from create responses. +use rand::{rngs::StdRng, Rng, SeedableRng}; +use serde::{Deserialize, Serialize}; + +/// Stream derivation constant for the data axis (topology gets its own). +const DATA_AXIS: u64 = 0x5eed_da7a_0000_0001; + +/// Access-control profile: share of creates that are protected (owned by the +/// owner identity) and share of protected docs that get a reader grant. +#[derive(Clone, Copy, Debug, PartialEq, Eq, Serialize, Deserialize)] +pub struct AcpProfile { + pub protected_pct: u32, + pub grant_pct: u32, +} + +/// Weight table over op kinds plus shape parameters. Weights, not percents. +#[derive(Clone, Debug, PartialEq, Serialize, Deserialize)] +pub struct Profile { + pub name: String, + pub create: u32, + pub update: u32, + pub delete: u32, + pub query: u32, + /// Target create payload size before store amplification. + pub doc_bytes: usize, + /// Ops per second on the virtual schedule. + pub rate: f64, + pub collection: String, + /// Fields named in `encryptFields:` on every create. Empty = plaintext profile. + #[serde(default)] + pub encrypt_fields: Vec, + /// Field with a searchable-encryption index; the query op becomes an SE query. + #[serde(default)] + pub se_field: Option, + /// Give every document its own `name` instead of drawing one of + /// `NAME_POOL`, so an SE query matches exactly one document. + #[serde(default)] + pub unique_names: bool, + /// Node indices that receive create ops; `None` = any node. Other ops are unaffected. + #[serde(default)] + pub create_nodes: Option>, + /// Weighted create-payload sizes as `(doc_bytes, weight)`. Empty means + /// "use `doc_bytes`", which is what every pre-existing profile does, and + /// is why their frozen plans cannot move: an empty mix draws no randomness. + #[serde(default)] + pub size_mix: Vec<(usize, u32)>, + /// `true` selects the Users schema with `@index` on `age`. Default false, + /// consumes no RNG, so existing frozen plans cannot move. + #[serde(default)] + pub indexed: bool, + /// One-to-many Author/Book profile. Default false, consumes no RNG. + #[serde(default)] + pub relation: bool, + #[serde(default)] + pub acp: Option, +} + +impl Profile { + /// The skeleton profile from the design (section 2). + pub fn p0_crud() -> Self { + Self { + name: "p0-crud".into(), + create: 30, + update: 40, + delete: 5, + query: 25, + doc_bytes: 1200, + rate: 20.0, + collection: "Users".into(), + encrypt_fields: Vec::new(), + se_field: None, + unique_names: false, + create_nodes: None, + size_mix: Vec::new(), + indexed: false, + relation: false, + acp: None, + } + } + + /// `p0-crud` with a within-run payload spread. The buckets are chosen so the + /// 1,200 B bucket is byte-identical to every published run's document, and the + /// largest stays below any candidate chunk threshold (design 152, reversal R1). + pub fn p0_size() -> Self { + Self { + name: "p0-size".into(), + size_mix: vec![(256, 40), (1_200, 30), (16_000, 20), (128_000, 10)], + ..Self::p0_crud() + } + } + + /// `p0-crud` at a fixed 256-byte create payload. Empty `size_mix`, so the + /// draw is `doc_bytes` only: a one-term disk comparison against `p0-crud` + /// and `p0-size-128k`, not a within-run mix. + pub fn p0_size_256() -> Self { + Self { + name: "p0-size-256".into(), + doc_bytes: 256, + ..Self::p0_crud() + } + } + + /// `p0-crud` at a fixed 128 KB create payload. Same empty-mix construction + /// as `p0_size_256`. Largest bucket stays below the candidate chunk + /// threshold (design 152, reversal R1). + pub fn p0_size_128k() -> Self { + Self { + name: "p0-size-128k".into(), + doc_bytes: 128_000, + ..Self::p0_crud() + } + } + + /// `p0-crud` with `@index` on `age`. The planned GraphQL is byte-identical + /// to p0-crud at the same seed; only the schema string differs. + pub fn p0_index() -> Self { + Self { + name: "p0-index".into(), + indexed: true, + ..Self::p0_crud() + } + } + + /// One-to-many Author/Book. Same op-kind weights as p0-crud; no size mix + /// and no index. Child creates name a parent *slot*; the executor fills + /// `author: ""`. + pub fn p3_relation() -> Self { + Self { + name: "p3-relation".into(), + collection: "Book".into(), + relation: true, + ..Self::p0_crud() + } + } + + /// M1b V1: encrypted `secret`/`pin`, SE index on `name` (spec 62, section 2). + pub fn p1_encrypted() -> Self { + Self { + name: "p1-encrypted".into(), + collection: "Vault".into(), + encrypt_fields: vec!["secret".into(), "pin".into()], + se_field: Some("name".into()), + ..Self::p0_crud() + } + } + + /// p1 with one document per name: same weights, fields and sizes, but the + /// SE query returns a single document, so a miss is the first-responder + /// defect and not the many-documents-per-name query shape (133 part 4, + /// rank 3). 307 remains the colliding-name result. + pub fn p1_unique() -> Self { + Self { + name: "p1-unique".into(), + unique_names: true, + ..Self::p1_encrypted() + } + } + + /// M1b V2: local document ACP, protected + public docs, reader grants (spec 64). + pub fn p2_acp() -> Self { + Self { + name: "p2-acp".into(), + collection: "User".into(), + acp: Some(AcpProfile { + protected_pct: 60, + grant_pct: 30, + }), + ..Self::p0_crud() + } + } + + pub fn by_name(name: &str) -> Option { + match name { + "p0-crud" => Some(Self::p0_crud()), + "p0-size" => Some(Self::p0_size()), + "p0-size-256" => Some(Self::p0_size_256()), + "p0-size-128k" => Some(Self::p0_size_128k()), + "p0-index" => Some(Self::p0_index()), + "p3-relation" => Some(Self::p3_relation()), + "p1-encrypted" => Some(Self::p1_encrypted()), + "p1-unique" => Some(Self::p1_unique()), + "p2-acp" => Some(Self::p2_acp()), + _ => None, + } + } + + /// Needs the ACP-enabled cluster (local ACP, owner/reader identities). + pub fn is_acp(&self) -> bool { + self.acp.is_some() + } + + /// Needs the encryption-enabled cluster (dev mode, identities, SE key). + pub fn is_encrypted(&self) -> bool { + !self.encrypt_fields.is_empty() || self.se_field.is_some() + } +} + +#[derive(Clone, Copy, Debug, PartialEq, Eq, Serialize, Deserialize)] +#[serde(rename_all = "snake_case")] +pub enum OpKind { + Create, + Update, + Delete, + Query, + Grant, +} + +/// Who issues an op on an ACP profile. `None` = the op has no identity (p0/p1). +#[derive(Clone, Copy, Debug, PartialEq, Eq, Hash, Serialize, Deserialize)] +#[serde(rename_all = "snake_case")] +pub enum Actor { + Owner, + Reader, + Anon, +} + +/// One planned op. +#[derive(Clone, Debug, PartialEq, Serialize, Deserialize)] +pub struct PlannedOp { + pub index: u64, + pub virtual_ts_ms: u64, + pub node: usize, + pub kind: OpKind, + /// Ledger slot for update/delete victims; the slot a create fills. + pub slot: Option, + /// GraphQL literal: create input object, update input object, or the + /// query selection. None for delete. + pub payload: Option, + /// SE query only: ledger slots whose name equals `payload` at planning + /// time; the executor resolves them to docIDs. Empty otherwise. + #[serde(default)] + pub expect_slots: Vec, + #[serde(default)] + pub actor: Option, + /// Per-op collection; `None` means the profile's single collection. + #[serde(default)] + pub collection: Option, + /// Child create: ledger slot of the parent Author. Never a docID. + #[serde(default)] + pub parent_slot: Option, +} + +pub const NAME_POOL: usize = 40; +/// One Author per this many Books, after the first create (always a parent). +const PARENT_EVERY: u32 = 8; + +pub struct Generator { + rng: StdRng, + profile: Profile, + nodes: usize, + next_index: u64, + /// Slots handed out so far (creation order). + created: usize, + /// Live slots; deletes swap_remove, so evolution depends only on the seed. + live: Vec, + /// Name per slot (creation order); only filled for encrypted profiles. + names: Vec, + /// Creator per slot (creation order); only filled for ACP profiles. + owner_of: Vec, + /// Create node per slot (creation order); only filled for ACP profiles. + /// Grants go there: the local DAC state lives on the node that served + /// the relationship add. + create_node: Vec, + /// Protected slots that already received a reader grant. + granted: std::collections::HashSet, + /// Live Author slots (relation profile only). + authors: Vec, + /// Live Book slots (relation profile only). + books: Vec, +} + +impl Generator { + pub fn new(seed: u64, profile: Profile, nodes: usize) -> Self { + Self { + rng: StdRng::seed_from_u64(seed ^ DATA_AXIS), + profile, + nodes, + next_index: 0, + created: 0, + live: Vec::new(), + names: Vec::new(), + owner_of: Vec::new(), + create_node: Vec::new(), + granted: std::collections::HashSet::new(), + authors: Vec::new(), + books: Vec::new(), + } + } + + pub fn next_op(&mut self) -> PlannedOp { + let (c, u, d, q, rate) = ( + self.profile.create, + self.profile.update, + self.profile.delete, + self.profile.query, + self.profile.rate, + ); + let acp = self.profile.acp; + let r = self.rng.gen_range(0..c + u + d + q); + let mut kind = if r < c { + OpKind::Create + } else if r < c + u { + // On ACP profiles the first `grant_pct` percent of the update band are grants. + match acp { + Some(a) if (r - c) * 100 < u * a.grant_pct => OpKind::Grant, + _ => OpKind::Update, + } + } else if r < c + u + d { + OpKind::Delete + } else { + OpKind::Query + }; + let grant_candidates: Vec = if kind == OpKind::Grant { + self.live + .iter() + .copied() + .filter(|s| { + self.owner_of.get(*s) == Some(&Actor::Owner) && !self.granted.contains(s) + }) + .collect() + } else { + Vec::new() + }; + // A victim op with nothing live becomes a create; still seed-determined. + if ((matches!(kind, OpKind::Update | OpKind::Delete) + || (kind == OpKind::Query && (self.profile.se_field.is_some() || acp.is_some()))) + && self.live.is_empty()) + || (kind == OpKind::Grant && grant_candidates.is_empty()) + { + kind = OpKind::Create; + } + let mut node = match (&kind, &self.profile.create_nodes) { + (OpKind::Create, Some(allowed)) if !allowed.is_empty() => { + allowed[self.rng.gen_range(0..allowed.len())] + } + _ => self.rng.gen_range(0..self.nodes), + }; + let mut expect_slots = Vec::new(); + let mut actor = None; + let mut collection = None; + let mut parent_slot = None; + let (slot, payload) = match kind { + OpKind::Create => { + let slot = self.created; + self.created += 1; + self.live.push(slot); + let payload = if self.profile.relation { + let is_parent = + self.authors.is_empty() || self.rng.gen_range(0..PARENT_EVERY + 1) == 0; + if is_parent { + self.authors.push(slot); + collection = Some("Author".into()); + self.create_author_input() + } else { + let pslot = self.authors[self.rng.gen_range(0..self.authors.len())]; + self.books.push(slot); + collection = Some("Book".into()); + parent_slot = Some(pslot); + self.create_book_input() + } + } else if self.profile.is_encrypted() { + let mut name = format!("name-{:02}", self.rng.gen_range(0..NAME_POOL)); + // p1-unique draws the same pool value, so both profiles + // plan the same ops from a seed; the slot suffix is what + // makes the name unique per document. + if self.profile.unique_names { + name = format!("{name}-{slot:06}"); + } + self.names.push(name.clone()); + self.create_input_vault(&name) + } else { + self.create_input() + }; + if let Some(a) = acp { + let protected = self.rng.gen_range(0..100) < a.protected_pct; + let who = if protected { Actor::Owner } else { Actor::Anon }; + self.owner_of.push(who); + self.create_node.push(node); + actor = Some(who); + } + (Some(slot), Some(payload)) + } + OpKind::Update => { + let pos = self.rng.gen_range(0..self.live.len()); + let slot = self.live[pos]; + let payload = if self.profile.relation { + if self.authors.contains(&slot) { + collection = Some("Author".into()); + self.update_author_input() + } else { + collection = Some("Book".into()); + self.update_book_input() + } + } else if self.profile.is_encrypted() { + self.update_input_vault() + } else { + self.update_input() + }; + if acp.is_some() { + actor = Some(self.owner_of[slot]); + } + (Some(slot), Some(payload)) + } + OpKind::Delete => { + let pos = self.rng.gen_range(0..self.live.len()); + let slot = self.live.swap_remove(pos); + if self.profile.relation { + if let Some(i) = self.authors.iter().position(|&s| s == slot) { + self.authors.swap_remove(i); + collection = Some("Author".into()); + } else { + if let Some(i) = self.books.iter().position(|&s| s == slot) { + self.books.swap_remove(i); + } + collection = Some("Book".into()); + } + } + if acp.is_some() { + actor = Some(self.owner_of[slot]); + } + (Some(slot), None) + } + OpKind::Grant => { + let slot = grant_candidates[self.rng.gen_range(0..grant_candidates.len())]; + self.granted.insert(slot); + node = self.create_node[slot]; + actor = Some(Actor::Owner); + (Some(slot), None) + } + // ACP query: the executor reads the slot as owner, reader and anon. + OpKind::Query if acp.is_some() => { + let pos = self.rng.gen_range(0..self.live.len()); + (Some(self.live[pos]), None) + } + OpKind::Query if self.profile.se_field.is_some() => { + let pos = self.rng.gen_range(0..self.live.len()); + let name = self.names[self.live[pos]].clone(); + expect_slots = self + .live + .iter() + .copied() + .filter(|s| self.names[*s] == name) + .collect(); + (None, Some(name)) + } + OpKind::Query => { + if self.profile.relation { + collection = Some("Book".into()); + } + (None, Some(self.query_selection())) + } + }; + let index = self.next_index; + self.next_index += 1; + PlannedOp { + index, + virtual_ts_ms: (index as f64 * 1000.0 / rate) as u64, + node, + kind, + slot, + payload, + expect_slots, + actor, + collection, + parent_slot, + } + } + + fn alnum(&mut self, len: usize) -> String { + const CHARS: &[u8] = b"abcdefghijklmnopqrstuvwxyz0123456789"; + (0..len) + .map(|_| CHARS[self.rng.gen_range(0..CHARS.len())] as char) + .collect() + } + + /// One RNG draw, and only when a mix exists. Profiles with an empty mix + /// never reach this, so their draw sequence is unchanged. + fn draw_doc_bytes(&mut self) -> usize { + let total: u32 = self.profile.size_mix.iter().map(|(_, w)| *w).sum(); + if total == 0 { + return self.profile.doc_bytes; + } + let mut pick = self.rng.gen_range(0..total); + for (bytes, weight) in &self.profile.size_mix { + if pick < *weight { + return *bytes; + } + pick -= *weight; + } + self.profile.size_mix[0].0 + } + + /// At most 8 significant digits, under the 15-digit float roundtrip + /// ceiling (a known runtime asymmetry). + fn score(&mut self) -> String { + format!("{:.2}", self.rng.gen_range(0..1_000_000) as f64 / 100.0) + } + + fn create_input(&mut self) -> String { + let name = self.alnum(8); + let age = self.rng.gen_range(0..100); + let score = self.score(); + let doc_bytes = if self.profile.size_mix.is_empty() { + self.profile.doc_bytes + } else { + self.draw_doc_bytes() + }; + let blob = self.alnum(doc_bytes.saturating_sub(64)); + format!("{{name: \"{name}\", age: {age}, score: {score}, blob: \"{blob}\"}}") + } + + fn update_input(&mut self) -> String { + let age = self.rng.gen_range(0..100); + let score = self.score(); + format!("{{age: {age}, score: {score}}}") + } + + fn create_input_vault(&mut self, name: &str) -> String { + let secret = self.alnum(16); + let pin = format!("{:04}", self.rng.gen_range(0..10_000)); + let score = self.score(); + let blob = self.alnum(self.profile.doc_bytes.saturating_sub(96)); + format!( + "{{name: \"{name}\", secret: \"{secret}\", pin: \"{pin}\", score: {score}, blob: \"{blob}\"}}" + ) + } + + fn update_input_vault(&mut self) -> String { + let secret = self.alnum(16); + let score = self.score(); + format!("{{secret: \"{secret}\", score: {score}}}") + } + + fn query_selection(&mut self) -> String { + if self.profile.relation { + let rating = self.rng.gen_range(0..100); + return format!( + "Book(limit: 10, filter: {{rating: {{_gt: {rating}}}}}) {{ _docID name rating }}" + ); + } + let age = self.rng.gen_range(0..100); + format!( + "{}(limit: 10, filter: {{age: {{_gt: {age}}}}}) {{ _docID name age }}", + self.profile.collection + ) + } + + fn create_author_input(&mut self) -> String { + let name = self.alnum(8); + let age = self.rng.gen_range(0..100); + let verified = self.rng.gen_range(0..2) == 1; + format!("{{name: \"{name}\", age: {age}, verified: {verified}}}") + } + + fn create_book_input(&mut self) -> String { + let name = self.alnum(8); + let rating = self.score(); + format!("{{name: \"{name}\", rating: {rating}}}") + } + + fn update_author_input(&mut self) -> String { + let age = self.rng.gen_range(0..100); + format!("{{age: {age}}}") + } + + fn update_book_input(&mut self) -> String { + let rating = self.score(); + format!("{{rating: {rating}}}") + } +} + +impl Iterator for Generator { + type Item = PlannedOp; + fn next(&mut self) -> Option { + Some(self.next_op()) + } +} + +#[cfg(test)] +mod tests { + use super::*; + + fn plan(seed: u64, n: usize) -> Vec { + Generator::new(seed, Profile::p0_crud(), 2) + .take(n) + .collect() + } + + #[test] + fn same_seed_same_sequence() { + assert_eq!(plan(42, 500), plan(42, 500)); + } + + #[test] + fn different_seed_different_sequence() { + assert_ne!(plan(42, 500), plan(43, 500)); + } + + /// Victims are always slots that were created earlier and not yet deleted. + #[test] + fn victims_are_live_slots() { + let mut created = 0usize; + let mut deleted = std::collections::HashSet::new(); + for op in plan(7, 2000) { + match op.kind { + OpKind::Create => { + assert_eq!(op.slot, Some(created)); + created += 1; + } + OpKind::Update | OpKind::Delete => { + let slot = op.slot.expect("victim slot"); + assert!(slot < created, "slot {slot} not created yet"); + assert!(!deleted.contains(&slot), "slot {slot} already deleted"); + if op.kind == OpKind::Delete { + deleted.insert(slot); + } + } + OpKind::Query => assert_eq!(op.slot, None), + OpKind::Grant => panic!("p0 plans no grants"), + } + } + assert!( + created > 0 && !deleted.is_empty(), + "profile must exercise all kinds" + ); + } + + #[test] + fn old_manifest_profile_still_loads() { + let old = r#"{"name":"p0-crud","create":30,"update":40,"delete":5,"query":25, + "doc_bytes":1200,"rate":5.0,"collection":"Users"}"#; + let p: Profile = serde_json::from_str(old).expect("old profile json"); + assert_eq!( + p, + Profile { + rate: 5.0, + ..Profile::p0_crud() + } + ); + assert!(p.encrypt_fields.is_empty() && p.se_field.is_none() && !p.is_encrypted()); + } + + #[test] + fn p1_encrypted_shape() { + let p = Profile::p1_encrypted(); + assert_eq!(p.collection, "Vault"); + assert_eq!( + p.encrypt_fields, + vec!["secret".to_string(), "pin".to_string()] + ); + assert_eq!(p.se_field.as_deref(), Some("name")); + assert!(p.is_encrypted()); + let back: Profile = serde_json::from_value(serde_json::to_value(&p).unwrap()).unwrap(); + assert_eq!(back, p); + } + + #[test] + fn profile_by_name() { + assert_eq!(Profile::by_name("p0-crud"), Some(Profile::p0_crud())); + assert_eq!( + Profile::by_name("p1-encrypted"), + Some(Profile::p1_encrypted()) + ); + assert_eq!( + Profile::by_name("p0-size-256"), + Some(Profile::p0_size_256()) + ); + assert_eq!( + Profile::by_name("p0-size-128k"), + Some(Profile::p0_size_128k()) + ); + assert_eq!(Profile::by_name("p0-index"), Some(Profile::p0_index())); + assert_eq!( + Profile::by_name("p3-relation"), + Some(Profile::p3_relation()) + ); + assert_eq!(Profile::by_name("nope"), None); + } + + /// p0 must not move: hash of the first 500 ops for seed 42, 2 nodes. + #[test] + fn p0_plan_is_frozen() { + use std::hash::{Hash, Hasher}; + let mut h = std::collections::hash_map::DefaultHasher::new(); + for op in plan(42, 500) { + ( + op.index, + op.virtual_ts_ms, + op.node, + op.kind as u8, + op.slot, + op.payload, + ) + .hash(&mut h); + } + assert_eq!(h.finish(), 12263656268250365760); + } + + fn plan_p1(seed: u64, n: usize) -> Vec { + Generator::new(seed, Profile::p1_encrypted(), 4) + .take(n) + .collect() + } + + fn plan_p0_size(seed: u64, n: usize) -> Vec { + Generator::new(seed, Profile::p0_size(), 2) + .take(n) + .collect() + } + + #[test] + fn p0_size_draws_more_than_one_payload_length() { + let lens: std::collections::BTreeSet = plan_p0_size(42, 500) + .into_iter() + .filter(|op| op.kind == OpKind::Create) + .map(|op| op.payload.unwrap().len()) + .collect(); + assert!( + lens.len() >= 3, + "expected a spread of create payload sizes, got {lens:?}" + ); + // age (1-2 digits) and score (4-7 chars) alone vary payload length by + // only a few bytes; a length this far past the 1200-byte bucket's + // ceiling (max realized length 1189) is reachable only by drawing the + // 16000 or 128000 byte buckets, so this fails if the size mix is not + // actually being exercised. + assert!( + lens.iter().any(|&l| l > 2000), + "expected a create payload only reachable via the 16000/128000 byte buckets, got {lens:?}" + ); + } + + #[test] + fn size_mix_is_empty_on_every_pre_existing_profile() { + for p in [ + Profile::p0_crud(), + Profile::p1_encrypted(), + Profile::p1_unique(), + Profile::p2_acp(), + ] { + assert!( + p.size_mix.is_empty(), + "{} must not carry a size mix", + p.name + ); + } + } + + /// p0-size must not move: hash of the first 500 ops for seed 42, 2 nodes. + #[test] + fn p0_size_plan_is_frozen() { + use std::hash::{Hash, Hasher}; + let mut h = std::collections::hash_map::DefaultHasher::new(); + for op in plan_p0_size(42, 500) { + ( + op.index, + op.virtual_ts_ms, + op.node, + op.kind as u8, + op.slot, + op.payload, + ) + .hash(&mut h); + } + assert_eq!(h.finish(), 11674702909862144291); + } + + fn plan_p0_size_256(seed: u64, n: usize) -> Vec { + Generator::new(seed, Profile::p0_size_256(), 2) + .take(n) + .collect() + } + + fn plan_p0_size_128k(seed: u64, n: usize) -> Vec { + Generator::new(seed, Profile::p0_size_128k(), 2) + .take(n) + .collect() + } + + #[test] + fn p0_size_256_is_a_single_small_bucket() { + let p = Profile::p0_size_256(); + assert_eq!(p.name, "p0-size-256"); + assert_eq!(p.doc_bytes, 256); + assert!( + p.size_mix.is_empty(), + "flat profiles use doc_bytes, not a mix" + ); + let lens: std::collections::BTreeSet = plan_p0_size_256(42, 200) + .into_iter() + .filter(|op| op.kind == OpKind::Create) + .map(|op| op.payload.unwrap().len()) + .collect(); + assert!( + lens.iter().all(|&l| (200..400).contains(&l)), + "expected GraphQL creates near 256 B, got {lens:?}" + ); + } + + #[test] + fn p0_size_128k_creates_are_large() { + let p = Profile::p0_size_128k(); + assert_eq!(p.name, "p0-size-128k"); + assert_eq!(p.doc_bytes, 128_000); + assert!( + p.size_mix.is_empty(), + "flat profiles use doc_bytes, not a mix" + ); + let lens: std::collections::BTreeSet = plan_p0_size_128k(42, 80) + .into_iter() + .filter(|op| op.kind == OpKind::Create) + .map(|op| op.payload.unwrap().len()) + .collect(); + assert!( + lens.iter().any(|&l| l > 100_000) && lens.iter().all(|&l| l < 130_000), + "expected GraphQL creates near 128 KB, got {lens:?}" + ); + } + + #[test] + fn p0_size_256_plan_is_frozen() { + use std::hash::{Hash, Hasher}; + let mut h = std::collections::hash_map::DefaultHasher::new(); + for op in plan_p0_size_256(42, 500) { + ( + op.index, + op.virtual_ts_ms, + op.node, + op.kind as u8, + op.slot, + op.payload, + ) + .hash(&mut h); + } + assert_eq!(h.finish(), 16343893043046409438); + } + + #[test] + fn p0_size_128k_plan_is_frozen() { + use std::hash::{Hash, Hasher}; + let mut h = std::collections::hash_map::DefaultHasher::new(); + for op in plan_p0_size_128k(42, 500) { + ( + op.index, + op.virtual_ts_ms, + op.node, + op.kind as u8, + op.slot, + op.payload, + ) + .hash(&mut h); + } + assert_eq!(h.finish(), 6215196610090915070); + } + + fn plan_p0_index(seed: u64, n: usize) -> Vec { + Generator::new(seed, Profile::p0_index(), 2) + .take(n) + .collect() + } + + #[test] + fn p0_index_is_p0_plus_a_flag() { + let p = Profile::p0_index(); + assert_eq!(p.name, "p0-index"); + assert!(p.indexed); + assert!(!Profile::p0_crud().indexed); + assert!(!Profile::p0_size().indexed); + assert!(p.size_mix.is_empty()); + assert_eq!(p.doc_bytes, Profile::p0_crud().doc_bytes); + } + + #[test] + fn p0_index_plan_equals_p0_plan() { + use std::hash::{Hash, Hasher}; + let tuple = |op: &PlannedOp| { + ( + op.index, + op.virtual_ts_ms, + op.node, + op.kind as u8, + op.slot, + op.payload.clone(), + ) + }; + let mut h0 = std::collections::hash_map::DefaultHasher::new(); + let mut hi = std::collections::hash_map::DefaultHasher::new(); + for op in plan(42, 500) { + tuple(&op).hash(&mut h0); + } + for op in plan_p0_index(42, 500) { + tuple(&op).hash(&mut hi); + } + assert_eq!(h0.finish(), 12263656268250365760); + assert_eq!(hi.finish(), h0.finish()); + } + + fn plan_p3(seed: u64, n: usize) -> Vec { + Generator::new(seed, Profile::p3_relation(), 2) + .take(n) + .collect() + } + + #[test] + fn p3_relation_is_unindexed_and_has_no_size_mix() { + let p = Profile::p3_relation(); + assert_eq!(p.name, "p3-relation"); + assert!(p.size_mix.is_empty()); + assert!(!p.indexed); + assert!(p.relation); + assert!(!Profile::p0_crud().relation); + } + + #[test] + fn p3_relation_first_create_is_an_author_without_a_parent_slot() { + let first = plan_p3(42, 20) + .into_iter() + .find(|op| op.kind == OpKind::Create) + .expect("a create"); + assert_eq!(first.collection.as_deref(), Some("Author")); + assert_eq!(first.parent_slot, None); + let p = first.payload.as_deref().unwrap(); + assert!(p.contains("name:"), "{p}"); + assert!(p.contains("age:"), "{p}"); + assert!(p.contains("verified:"), "{p}"); + assert!(!p.contains("author:"), "{p}"); + assert!(!p.contains("blob:"), "{p}"); + } + + #[test] + fn p3_relation_child_carries_a_parent_slot_not_a_doc_id() { + let ops = plan_p3(42, 400); + let child = ops + .iter() + .find(|op| op.kind == OpKind::Create && op.collection.as_deref() == Some("Book")) + .expect("a Book create"); + let parent_slot = child.parent_slot.expect("child must name a parent slot"); + let parent = ops + .iter() + .find(|op| op.kind == OpKind::Create && op.slot == Some(parent_slot)) + .expect("parent create at that slot"); + assert_eq!(parent.collection.as_deref(), Some("Author")); + let payload = child.payload.as_deref().unwrap(); + assert!(payload.contains("name:"), "{payload}"); + assert!(payload.contains("rating:"), "{payload}"); + assert!( + !payload.contains("author:"), + "docID is resolved at execute time, got {payload}" + ); + assert!(!payload.contains("bae-"), "{payload}"); + } + + #[test] + fn p0_plans_do_not_set_relation_fields() { + for op in plan(42, 200) { + assert!(op.collection.is_none()); + assert!(op.parent_slot.is_none()); + } + } + + #[test] + fn p3_relation_plan_is_frozen() { + use std::hash::{Hash, Hasher}; + let mut h = std::collections::hash_map::DefaultHasher::new(); + for op in plan_p3(42, 500) { + ( + op.index, + op.virtual_ts_ms, + op.node, + op.kind as u8, + op.slot, + op.payload.clone(), + op.collection.clone(), + op.parent_slot, + ) + .hash(&mut h); + } + assert_eq!(h.finish(), 1928505163324143662); + } + + #[test] + fn p0_ops_have_no_expect_slots() { + assert!(plan(42, 500).iter().all(|op| op.expect_slots.is_empty())); + } + + #[test] + fn p1_create_and_update_payloads() { + let ops = plan_p1(9, 400); + let create = ops.iter().find(|o| o.kind == OpKind::Create).unwrap(); + let p = create.payload.as_deref().unwrap(); + assert!(p.starts_with("{name: \"") && p.contains("secret: \"") && p.contains("pin: \"")); + assert!(p.contains("score: ") && p.contains("blob: \"") && !p.contains("age:")); + let update = ops.iter().find(|o| o.kind == OpKind::Update).unwrap(); + let u = update.payload.as_deref().unwrap(); + assert!(u.starts_with("{secret: \"") && u.contains("score: ") && !u.contains("age:")); + } + + /// An SE query names a live doc's name and lists every live slot with that name. + #[test] + fn p1_query_carries_expected_live_slots() { + let mut names: std::collections::HashMap = Default::default(); + let mut live = std::collections::HashSet::new(); + let mut queries = 0; + for op in plan_p1(11, 3000) { + match op.kind { + OpKind::Create => { + let slot = op.slot.unwrap(); + let p = op.payload.as_deref().unwrap(); + let name = p["{name: \"".len()..] + .split('"') + .next() + .unwrap() + .to_string(); + names.insert(slot, name); + live.insert(slot); + } + OpKind::Delete => { + live.remove(&op.slot.unwrap()); + } + OpKind::Query => { + queries += 1; + let name = op.payload.as_deref().expect("SE query payload is the name"); + let mut want: Vec = + live.iter().copied().filter(|s| names[s] == name).collect(); + want.sort(); + let mut got = op.expect_slots.clone(); + got.sort(); + assert_eq!(got, want, "op {}", op.index); + assert!(!got.is_empty()); + } + OpKind::Update => {} + OpKind::Grant => panic!("p1 plans no grants"), + } + } + assert!(queries > 50, "profile must exercise SE queries"); + } + + #[test] + fn p1_names_come_from_a_small_pool() { + let names: std::collections::HashSet = plan_p1(3, 2000) + .into_iter() + .filter(|o| o.kind == OpKind::Create) + .map(|o| { + o.payload.unwrap()["{name: \"".len()..] + .split('"') + .next() + .unwrap() + .to_string() + }) + .collect(); + assert!(names.len() <= NAME_POOL && names.len() > 10); + } + + #[test] + fn create_nodes_restricts_creates_only() { + let mut p = Profile::p1_encrypted(); + p.create_nodes = Some(vec![0, 1]); + let ops: Vec = Generator::new(21, p, 4).take(2000).collect(); + assert!(ops + .iter() + .filter(|o| o.kind == OpKind::Create) + .all(|o| o.node < 2)); + assert!( + ops.iter().any(|o| o.kind != OpKind::Create && o.node >= 2), + "other ops still reach Go nodes" + ); + } + + #[test] + fn create_nodes_absent_keeps_p0_frozen_and_loads_old_json() { + let old = r#"{"name":"p0-crud","create":30,"update":40,"delete":5,"query":25, + "doc_bytes":1200,"rate":5.0,"collection":"Users"}"#; + let p: Profile = serde_json::from_str(old).unwrap(); + assert_eq!(p.create_nodes, None); + } + + fn plan_p1_frozen_input() -> Vec { + Generator::new(42, Profile::p1_encrypted(), 4) + .take(500) + .collect() + } + + /// p1 must not move either: hash of the first 500 ops for seed 42, 4 nodes. + #[test] + fn p1_plan_is_frozen() { + use std::hash::{Hash, Hasher}; + let mut h = std::collections::hash_map::DefaultHasher::new(); + for op in plan_p1_frozen_input() { + ( + op.index, + op.virtual_ts_ms, + op.node, + op.kind as u8, + op.slot, + op.payload, + op.expect_slots, + ) + .hash(&mut h); + } + assert_eq!(h.finish(), 133609464861947189); + } + + fn plan_p1_unique(seed: u64, n: usize) -> Vec { + Generator::new(seed, Profile::p1_unique(), 4) + .take(n) + .collect() + } + + /// p1-unique must not move either: hash of the first 500 ops for seed 42, + /// 4 nodes, same shape as the p0 and p1 locks. + #[test] + fn p1_unique_plan_is_frozen() { + use std::hash::{Hash, Hasher}; + let mut h = std::collections::hash_map::DefaultHasher::new(); + for op in plan_p1_unique(42, 500) { + ( + op.index, + op.virtual_ts_ms, + op.node, + op.kind as u8, + op.slot, + op.payload, + op.expect_slots, + ) + .hash(&mut h); + } + assert_eq!(h.finish(), 18060237726193507568); + } + + /// The point of the profile: over a plan the size of run 307 (seed 307, + /// 4 nodes, 5401 ops executed) no two documents share a name, so an SE + /// query has exactly one right answer. + #[test] + fn p1_unique_never_repeats_a_name() { + let names: Vec = plan_p1_unique(307, 5401) + .into_iter() + .filter(|o| o.kind == OpKind::Create) + .map(|o| { + o.payload.unwrap()["{name: \"".len()..] + .split('"') + .next() + .unwrap() + .to_string() + }) + .collect(); + assert!( + names.len() > 1000, + "expected 307-sized creates, got {}", + names.len() + ); + let unique: std::collections::HashSet<&String> = names.iter().collect(); + assert_eq!(unique.len(), names.len(), "p1-unique repeated a name"); + // Every SE query therefore expects exactly one slot. + assert!(plan_p1_unique(307, 5401) + .iter() + .filter(|o| o.kind == OpKind::Query) + .all(|o| o.expect_slots.len() == 1)); + } + + /// Identical to p1 apart from the names, and it is still an SE profile. + #[test] + fn p1_unique_matches_p1_except_names() { + let (u, p1) = (Profile::p1_unique(), Profile::p1_encrypted()); + assert_eq!( + (u.create, u.update, u.delete, u.query, u.doc_bytes, u.rate), + ( + p1.create, + p1.update, + p1.delete, + p1.query, + p1.doc_bytes, + p1.rate + ) + ); + assert_eq!(u.collection, p1.collection); + assert_eq!(u.encrypt_fields, p1.encrypt_fields); + assert_eq!(u.se_field, p1.se_field); + assert!(u.unique_names && !p1.unique_names); + assert!(u.is_encrypted() && !u.is_acp()); + assert_eq!(Profile::by_name("p1-unique"), Some(u.clone())); + let back: Profile = serde_json::from_value(serde_json::to_value(&u).unwrap()).unwrap(); + assert_eq!(back, u); + } + + #[test] + fn old_manifest_profile_has_no_acp() { + let old = r#"{"name":"p1-encrypted","create":30,"update":40,"delete":5,"query":25,"doc_bytes":1200, + "rate":3.0,"collection":"Vault","encrypt_fields":["secret","pin"],"se_field":"name"}"#; + let p: Profile = serde_json::from_str(old).unwrap(); + assert!(p.acp.is_none() && !p.is_acp()); + } + + #[test] + fn p2_acp_shape() { + let p = Profile::p2_acp(); + assert_eq!(p.collection, "User"); + assert_eq!( + p.acp, + Some(AcpProfile { + protected_pct: 60, + grant_pct: 30 + }) + ); + assert!(p.is_acp() && !p.is_encrypted()); + assert_eq!(Profile::by_name("p2-acp"), Some(p.clone())); + let back: Profile = serde_json::from_value(serde_json::to_value(&p).unwrap()).unwrap(); + assert_eq!(back, p); + } + + #[test] + fn planned_op_actor_defaults_to_none() { + let v = serde_json::json!({"index":0,"virtual_ts_ms":0,"node":0,"kind":"create","slot":0,"payload":"{}"}); + let op: PlannedOp = serde_json::from_value(v).unwrap(); + assert_eq!(op.actor, None); + assert_eq!( + serde_json::to_value(Actor::Reader).unwrap(), + serde_json::json!("reader") + ); + assert_eq!( + serde_json::to_value(OpKind::Grant).unwrap(), + serde_json::json!("grant") + ); + } + + fn plan_p2(seed: u64, n: usize) -> Vec { + Generator::new(seed, Profile::p2_acp(), 4).take(n).collect() + } + + #[test] + fn p2_actors_and_grants() { + let ops = plan_p2(5, 3000); + let mut owner: std::collections::HashMap = Default::default(); + let mut create_node: std::collections::HashMap = Default::default(); + let mut granted = std::collections::HashSet::new(); + let (mut prot, mut pub_, mut grants, mut queries) = (0, 0, 0, 0); + for op in &ops { + match op.kind { + OpKind::Create => { + let a = op.actor.expect("create has an actor"); + assert!(matches!(a, Actor::Owner | Actor::Anon)); + if a == Actor::Owner { + prot += 1 + } else { + pub_ += 1 + } + owner.insert(op.slot.unwrap(), a); + create_node.insert(op.slot.unwrap(), op.node); + assert!(op.payload.as_deref().unwrap().starts_with("{name: \"")); + } + OpKind::Update | OpKind::Delete => { + assert_eq!(op.actor, Some(owner[&op.slot.unwrap()]), "op {}", op.index); + } + OpKind::Grant => { + grants += 1; + let s = op.slot.unwrap(); + assert_eq!(owner[&s], Actor::Owner, "grants only on protected docs"); + assert!(granted.insert(s), "slot {s} granted twice"); + assert_eq!(op.actor, Some(Actor::Owner)); + assert_eq!(op.node, create_node[&s], "grant on the origin node"); + } + OpKind::Query => { + queries += 1; + assert!(op.slot.is_some() && op.actor.is_none() && op.payload.is_none()); + } + } + } + let share = prot as f64 / (prot + pub_) as f64; + assert!((0.5..0.7).contains(&share), "protected share {share}"); + assert!(grants > 20 && queries > 100); + } + + #[test] + fn p0_and_p1_have_no_actor() { + assert!(plan(42, 300) + .iter() + .all(|o| o.actor.is_none() && o.kind != OpKind::Grant)); + assert!(plan_p1_frozen_input().iter().all(|o| o.actor.is_none())); + } +} diff --git a/crates/soak/src/main.rs b/crates/soak/src/main.rs new file mode 100644 index 0000000..f0f8e1f --- /dev/null +++ b/crates/soak/src/main.rs @@ -0,0 +1,1460 @@ +//! `soak`: cross-runtime DefraDB soak driver (M0 skeleton). +//! +//! Boots a mixed Go/Rust mesh through `defra-harness`, drives a seeded +//! workload, checks convergence between the runtimes, and writes a replayable +//! run artifact under `runs/-/`. +//! +//! ```text +//! soak run [--seed N] [--ops N] [--secs S] [--rate OPS_PER_SEC] [--control] +//! [--churn [--churn-spacing SECS]] [--grace SECS] [--settle SECS] +//! [--ceiling-mb MB] [--floor-rate R] [--meter-secs S] +//! [--min-settle SECS] [--no-subscribe] [--node-env KEY=VALUE]... +//! [--retry-intervals 5,10,20,40] [--sse-go] [--until-op N] [--hold] +//! [--nodes process|docker] [--reuse-network] [--topology 2r2g] +//! soak replay --manifest /manifest.json [--until-op N] [--hold] +//! [--grace SECS] [--settle SECS] +//! soak summarize +//! soak compare +//! soak manage --topology 2r0g --out [--cases R2,A2,S1] [--transport libp2p|iroh] +//! [--node-env KEY=VALUE]... +//! ``` +//! Env: `DEFRA_RUST_BINARY` (built `defra`), Go `defradb` on PATH with +//! `DEFRA_GO_COMPAT_COMMIT` set. +//! +//! `--control` adds a `Control` collection replicated Rust -> Go only and +//! writes to the Go side, so the checker must report M1 and M3 divergences +//! on it (the positive control). `--churn` enables the seeded restart / +//! crash-kill schedule. `replay` rebuilds a run from its manifest: same +//! seed, profile, executed op count and churn schedule, no disk budget; +//! `--until-op` stops the workload early and `--hold` keeps the mesh up for +//! inspection until Enter. `--retry-intervals` sets both runtimes' +//! `--replicator-retry-intervals` (default ladder 30,60,120,240,480,960,1920 +//! s) and is recorded in the manifest, since it changes the system under +//! test; both backends take it. `--min-settle SECS` keeps the settle (and +//! the meter) running that long after the workload stops even if the mesh is +//! already clear, which is the only way a run gets an idle sample. +//! `--no-subscribe` skips the collection subscribe, leaving the replicator as +//! the only delivery path, so a push failure is visible instead of masked by +//! gossip. `--node-env KEY=VALUE`, repeatable, reaches every node on both +//! backends and both runtimes, and is recorded in the manifest; it is how a +//! run raises its log level (`--node-env RUST_LOG=debug`). `compare` checks two runs against the replay +//! contract: planned op fields and the churn schedule, plus docIDs where +//! both runs have one; outcomes and timing are not part of it. +//! +//! `--nodes docker` runs the M2 six-node topology as containers on a +//! `soak-` network instead of harness processes (p0-crud only); +//! the backend is recorded in the manifest and honoured by `replay`. +//! +//! `manage` runs the management-channel cases (`manage/cases.rs`) on a +//! NAC-enabled Rust mesh and writes `summary.json` + `cases.md` under +//! `--out`; `--cases` selects from the table, default all. + +mod auth; +mod checker; +mod churn; +mod confirm; +mod executor; +mod generator; +mod manage; +mod meter; +mod nodes; +mod sse; +mod summary; +mod tags; + +use std::collections::HashMap; +use std::path::{Path, PathBuf}; +use std::process::Command; +use std::sync::atomic::{AtomicBool, AtomicU64, Ordering}; +use std::sync::Arc; +use std::time::{Duration, Instant, SystemTime, UNIX_EPOCH}; + +use defra_harness::{BinarySource, TestCluster}; +use eyre::{bail, eyre, Result, WrapErr}; +use serde_json::{json, Value}; +use tokio::sync::{mpsc, oneshot}; + +use auth::{auth_token, Identities, Identity}; +use checker::{Checker, CheckerConfig, Touch}; +use churn::{ChurnClock, ChurnConfig, ChurnEvent}; +use executor::{gql, gql_as, http_client, now_ms, Executor}; +use generator::{Actor, Generator, OpKind, Profile}; +use meter::{Meter, MeterConfig}; +use nodes::{DockerNodes, Nodes}; + +const SCHEMA: &str = "type Users { name: String age: Int score: Float blob: String }"; +const INDEXED_SCHEMA: &str = + "type Users { name: String age: Int @index score: Float blob: String }"; +const RELATION_SCHEMA: &str = "type Book {\n name: String\n rating: Float\n author: Author\n}\n\ntype Author {\n name: String\n age: Int\n verified: Boolean\n published: [Book]\n}"; +const VAULT_SCHEMA: &str = + "type Vault { name: String secret: String pin: String score: Float blob: String }"; +/// Shared searchable-encryption key for every node; the key is not under test. +const SE_KEY: [u8; 32] = [0x5e; 32]; + +fn schema_for(profile: &Profile) -> &'static str { + if profile.is_encrypted() { + VAULT_SCHEMA + } else if profile.relation { + RELATION_SCHEMA + } else if profile.indexed { + INDEXED_SCHEMA + } else { + SCHEMA + } +} + +fn mesh_collections(profile: &Profile) -> Vec { + if profile.relation { + vec!["Author".into(), "Book".into()] + } else { + vec![profile.collection.clone()] + } +} + +/// The p2-acp collection, bound to the policy added at setup. +fn acp_schema(policy_id: &str) -> String { + format!( + "type User @policy(id: \"{policy_id}\", resource: \"users\") {{ name: String age: Int score: Float blob: String }}" + ) +} + +/// Mesh shape from `--topology rg`: `rust` Rust nodes, then `go` Go ones. +/// Either count may be zero, which is the single-runtime control for a mixed +/// mesh; both zero is refused. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +struct Topology { + rust: usize, + go: usize, +} + +impl Topology { + fn parse(s: &str) -> Result { + let shape = "--topology must look like 2r2g"; + let count = |v: &str| -> Result { + eyre::ensure!( + !v.is_empty() && v.bytes().all(|b| b.is_ascii_digit()), + shape + ); + v.parse().wrap_err(shape) + }; + let (rust, rest) = s.split_once('r').ok_or_else(|| eyre!(shape))?; + let go = rest.strip_suffix('g').ok_or_else(|| eyre!(shape))?; + let t = Self { + rust: count(rust)?, + go: count(go)?, + }; + eyre::ensure!(t.total() > 0, "--topology needs at least one node"); + Ok(t) + } + + fn total(self) -> usize { + self.rust + self.go + } + + fn label(self) -> String { + format!("{}r{}g", self.rust, self.go) + } +} + +/// The Rust nodes' P2P transport, `--transport libp2p|iroh` (`manage`). +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +enum Transport { + Libp2p, + Iroh, +} + +impl Transport { + fn parse(s: Option<&str>) -> Result { + match s { + None | Some("libp2p") => Ok(Self::Libp2p), + Some("iroh") => Ok(Self::Iroh), + Some(other) => bail!("unknown --transport {other}; use libp2p or iroh"), + } + } + + fn label(self) -> &'static str { + match self { + Self::Libp2p => "libp2p", + Self::Iroh => "iroh", + } + } +} + +/// Kind and store per node index. Without `--topology` each backend keeps the +/// shape every published run used: the M2 six in containers, two of each as +/// processes. +fn node_specs(topology: Option, docker: bool) -> Vec { + match (topology, docker) { + (Some(t), _) => nodes::topology_specs(t.rust, t.go), + (None, true) => nodes::m2_specs(), + (None, false) => nodes::topology_specs(2, 2), + } +} + +fn generate_identity(bin: &Path, what: &str) -> Result { + let id = defra_harness::identity::generate_identity(bin) + .wrap_err_with(|| format!("generating the {what} identity"))?; + Ok(Identity { + key_hex: id.private_key_hex, + did: id.did, + }) +} +const CONTROL: &str = "Control"; +const CONTROL_SCHEMA: &str = "type Control { v: Int }"; +/// Container images for `--nodes docker`, tagged by the commit they hold. +const RUST_IMAGE: &str = "soak-defra:8d8bb299f"; + +/// Everything a run needs; `replay` rebuilds it from a manifest. +struct RunArgs { + seed: u64, + ops: usize, + secs: Option, + profile: Profile, + control: bool, + churn: Option, + /// The original's stored schedule on replay; `None` draws one. + churn_schedule: Option>, + grace: Duration, + settle: Duration, + /// Settle at least this long after the workload stops, clear or not. + min_settle: Duration, + ceiling_bytes: u64, + floor_rate: f64, + meter_interval: Duration, + until_op: Option, + hold: bool, + replay_of: Option, + /// Comma-separated seconds for both nodes' replicator retry ladder. + retry_intervals: Option, + /// Also subscribe on Go nodes (reproduces the memory growth). + sse_go: bool, + /// Skip the collection subscribe: replicators are the only delivery path. + no_subscribe: bool, + /// `KEY=VALUE` pairs set on every node, both backends and both runtimes. + node_env: Vec, + /// Containers on a `soak-` network instead of harness processes. + docker: bool, + /// Start even if a `soak-*` network is left over from an earlier run. + reuse_network: bool, + topology: Option, + /// `--node-acp-enable` with a startup identity on every node (`manage`). + nac: bool, + transport: Transport, +} + +impl RunArgs { + fn from_flags() -> Result { + let profile_name = flag("profile").unwrap_or_else(|| "p0-crud".to_string()); + let mut profile = Profile::by_name(&profile_name).ok_or_else(|| { + eyre!( + "unknown --profile {profile_name}; use p0-crud, p0-size, p0-size-256, p0-size-128k, p0-index, p3-relation, p1-encrypted, p1-unique or p2-acp" + ) + })?; + if let Some(rate) = flag("rate") { + profile.rate = rate.parse().wrap_err("--rate must be a number")?; + } + let docker = match flag("nodes").as_deref() { + None | Some("process") => false, + Some("docker") => true, + Some(other) => bail!("unknown --nodes {other}; use process or docker"), + }; + let topology = match flag("topology") { + Some(spec) => Some(Topology::parse(&spec)?), + None => None, + }; + let node_count = node_specs(topology, docker).len(); + if let Some(list) = flag("create-nodes") { + let nodes: Vec = list + .split(',') + .map(|s| { + s.trim() + .parse::() + .wrap_err("--create-nodes must be node indices") + }) + .collect::>()?; + eyre::ensure!( + nodes.iter().all(|n| *n < node_count), + "--create-nodes: index out of range" + ); + profile.create_nodes = Some(nodes); + } + let mut churn = has_flag("churn").then(|| ChurnConfig { + clock: ChurnClock::Wall, + ..ChurnConfig::default() + }); + if let (Some(cfg), Some(secs)) = (churn.as_mut(), flag("churn-spacing")) { + let ms = secs + .parse::() + .wrap_err("--churn-spacing must be seconds")? + * 1000; + cfg.spacing_ms = ms; + cfg.cooldown_ms = ms; + } + let checker = CheckerConfig::default(); + Ok(Self { + seed: parse_flag("seed", unix_secs())?, + ops: parse_flag("ops", 200)?, + secs: opt_flag("secs")?, + profile, + control: has_flag("control"), + churn, + churn_schedule: None, + grace: Duration::from_secs(parse_flag("grace", checker.grace.as_secs())?), + settle: Duration::from_secs(parse_flag("settle", checker.settle.as_secs())?), + min_settle: Duration::from_secs(parse_flag( + "min-settle", + checker.min_settle.as_secs(), + )?), + ceiling_bytes: (parse_flag::("ceiling-mb", 120.0 * 1024.0)? * 1_048_576.0) as u64, + floor_rate: parse_flag("floor-rate", 0.5)?, + meter_interval: Duration::from_secs(parse_flag("meter-secs", 60)?), + until_op: opt_flag("until-op")?, + hold: has_flag("hold"), + replay_of: None, + retry_intervals: flag("retry-intervals"), + sse_go: has_flag("sse-go"), + no_subscribe: has_flag("no-subscribe"), + node_env: { + let items = flags("node-env"); + for kv in &items { + split_node_env(kv)?; + } + items + }, + docker, + reuse_network: has_flag("reuse-network"), + topology, + nac: false, + transport: Transport::Libp2p, + }) + } + + /// The original run's parameters. The stored churn schedule is replayed + /// as recorded (older manifests without one are redrawn from the seed + /// over the planned `ops`); the workload stops at the original's + /// executed count via `until_op`, and the disk budget is off. + fn from_manifest(path: &Path) -> Result { + let m: Value = serde_json::from_str( + &std::fs::read_to_string(path) + .wrap_err_with(|| format!("reading {}", path.display()))?, + )?; + let caps = &m["caps"]; + let churn = match m["churn"]["config"].as_object() { + Some(_) => Some(serde_json::from_value(m["churn"]["config"].clone())?), + None => None, + }; + let churn_schedule = match m["churn"]["schedule"].as_array() { + Some(_) => Some(serde_json::from_value(m["churn"]["schedule"].clone())?), + None => None, + }; + Ok(Self { + seed: m["seed"] + .as_u64() + .ok_or_else(|| eyre!("manifest has no seed"))?, + ops: m["ops"] + .as_u64() + .ok_or_else(|| eyre!("manifest has no ops"))? as usize, + secs: None, + profile: serde_json::from_value(m["profile"].clone()).wrap_err("manifest profile")?, + control: m["control"].as_bool().unwrap_or(false), + churn, + churn_schedule, + grace: Duration::from_secs(parse_flag( + "grace", + caps["grace_secs"].as_u64().unwrap_or(120), + )?), + settle: Duration::from_secs(parse_flag( + "settle", + caps["settle_secs"].as_u64().unwrap_or(120), + )?), + min_settle: Duration::from_secs(parse_flag( + "min-settle", + caps["min_settle_secs"].as_u64().unwrap_or(0), + )?), + ceiling_bytes: u64::MAX / 2, + floor_rate: caps["floor_rate"].as_f64().unwrap_or(0.5), + meter_interval: Duration::from_secs(caps["meter_secs"].as_u64().unwrap_or(60)), + until_op: opt_flag("until-op")?.or_else(|| m["ops_executed"].as_u64()), + hold: has_flag("hold"), + replay_of: m["run_id"].as_str().map(String::from), + retry_intervals: caps["retry_intervals"].as_str().map(String::from), + sse_go: has_flag("sse-go") || caps["sse_go"].as_bool().unwrap_or(false), + no_subscribe: has_flag("no-subscribe") + || caps["no_subscribe"].as_bool().unwrap_or(false), + node_env: match flags("node-env") { + items if !items.is_empty() => items, + _ => caps["node_env"] + .as_array() + .into_iter() + .flatten() + .filter_map(|v| v.as_str().map(String::from)) + .collect(), + }, + docker: m["nodes"][0]["backend"] == json!("docker"), + topology: match m["topology"].as_str() { + Some(spec) => Some(Topology::parse(spec)?), + None => None, + }, + reuse_network: has_flag("reuse-network"), + nac: false, + transport: Transport::Libp2p, + }) + } +} + +fn main() -> Result<()> { + let argv: Vec = std::env::args().skip(1).collect(); + let cmd = argv + .first() + .map(String::as_str) + .filter(|a| !a.starts_with("--")) + .unwrap_or("run"); + match cmd { + "summarize" => { + let dir = argv + .get(1) + .ok_or_else(|| eyre!("usage: soak summarize "))?; + summary::write_profile(Path::new(dir))?; + print!( + "{}", + std::fs::read_to_string(Path::new(dir).join("profile.md"))? + ); + Ok(()) + } + "compare" => { + let (a, b) = match (argv.get(1), argv.get(2)) { + (Some(a), Some(b)) => (a, b), + _ => eyre::bail!("usage: soak compare "), + }; + let report = summary::compare(Path::new(a), Path::new(b))?; + println!("{}", serde_json::to_string_pretty(&report)?); + eyre::ensure!( + report["identical"] == json!(true) || report["prefix_identical"] == json!(true), + "runs differ on the replay contract" + ); + Ok(()) + } + "run" | "replay" => { + eyre::ensure!( + std::env::var_os("DEFRA_RUST_BINARY").is_some(), + "set DEFRA_RUST_BINARY to a built `defra` (e.g. /target/debug/defra)" + ); + let args = if cmd == "replay" { + let m = flag("manifest").ok_or_else(|| eyre!("replay needs --manifest "))?; + RunArgs::from_manifest(Path::new(&m))? + } else { + RunArgs::from_flags()? + }; + let run_dir = new_run_dir(args.seed)?; + // defra-harness puts node dirs under /target/e2e and + // deletes them on drop. Point the workspace at the run dir and + // keep them so the artifact holds the node data and logs. Set + // before any thread exists. + std::env::set_var("DEFRA_WORKSPACE_ROOT", &run_dir); + std::env::set_var("DEFRA_E2E_KEEP", "1"); + println!( + "run dir: {} seed: {} ops: {}{}", + run_dir.display(), + args.seed, + args.ops, + args.replay_of + .as_deref() + .map(|r| format!(" (replay of {r})")) + .unwrap_or_default() + ); + tokio::runtime::Runtime::new()?.block_on(run(&run_dir, args)) + } + "manage" => { + eyre::ensure!( + std::env::var_os("DEFRA_RUST_BINARY").is_some(), + "set DEFRA_RUST_BINARY to a built `defra` (e.g. /target/debug/defra)" + ); + let mut args = RunArgs::from_flags()?; + args.nac = true; + args.transport = Transport::parse(flag("transport").as_deref())?; + let out = flag("out").ok_or_else(|| eyre!("manage needs --out "))?; + std::fs::create_dir_all(&out)?; + let out = Path::new(&out).canonicalize()?; + std::env::set_var("DEFRA_WORKSPACE_ROOT", &out); + std::env::set_var("DEFRA_E2E_KEEP", "1"); + println!("out dir: {}", out.display()); + tokio::runtime::Runtime::new()?.block_on(manage::run(&out, args)) + } + other => { + eyre::bail!("unknown command {other}; use run, replay, manage, summarize or compare") + } + } +} + +async fn run(run_dir: &Path, a: RunArgs) -> Result<()> { + let mut nodes = start_nodes(run_dir, &a).await?; + let result = drive(run_dir, &a, &mut nodes).await; + let shutdown = nodes.shutdown().await; + result?; + shutdown?; + summary::write_profile(run_dir)?; + println!("profile: {}", run_dir.join("profile.md").display()); + Ok(()) +} + +/// Harness processes, or the M2 six-node topology as containers. +async fn start_nodes(run_dir: &Path, a: &RunArgs) -> Result { + let specs = node_specs(a.topology, a.docker); + println!( + "topology: {} node(s), {}", + specs.len(), + specs + .iter() + .map(|s| s.name.as_str()) + .collect::>() + .join(" ") + ); + if a.docker { + eyre::ensure!( + !a.profile.is_encrypted() && !a.profile.is_acp(), + "docker backend supports p0-crud only in M2 stage one" + ); + // The host CLIs must resolve before any container exists: a panic + // in `client()` would skip the teardown. Only the runtimes this + // topology actually runs are required. + for kind in [nodes::NodeKind::Rust, nodes::NodeKind::Go] { + if specs.iter().any(|s| s.kind == kind) { + kind.host_binary()?; + } + } + let leftover = nodes::existing_networks().await?; + eyre::ensure!( + leftover.is_empty() || a.reuse_network, + "soak networks exist: {}; remove them (docker network rm) or pass --reuse-network", + leftover.join(", ") + ); + let run_id = run_dir + .file_name() + .map(|n| n.to_string_lossy().into_owned()) + .unwrap_or_default(); + // The Go image tag is the compat commit, so image and host binary + // always name the same version. + let commit = std::env::var("DEFRA_GO_COMPAT_COMMIT") + .wrap_err("DEFRA_GO_COMPAT_COMMIT must be set for the docker backend")?; + eyre::ensure!( + !commit.is_empty(), + "DEFRA_GO_COMPAT_COMMIT must be set for the docker backend" + ); + let go_image = format!("soak-defradb:{commit}"); + let images = (RUST_IMAGE, go_image.as_str()); + if let Some(intervals) = &a.retry_intervals { + println!("replicator retry intervals in every container: {intervals}s"); + } + let docker = DockerNodes::start( + &run_id, + run_dir, + specs, + images, + a.retry_intervals.as_deref(), + &a.node_env, + ) + .await + .wrap_err("starting the docker cluster")?; + println!("docker backend: {RUST_IMAGE} + {go_image} on soak-{run_id}"); + return Ok(Nodes::Docker(docker)); + } + // The harness spawns nodes as children of this process and does not clear + // their environment, so setting it here reaches every node of either + // runtime, including the ones a churn event respawns. + apply_node_env(&a.node_env)?; + let rust_n = specs + .iter() + .filter(|s| s.kind == nodes::NodeKind::Rust) + .count(); + let mut builder = TestCluster::builder() + .rust_nodes(rust_n) + .go_nodes(specs.len() - rust_n) + .with_p2p() + // File keyrings so peer identities survive restarts: without one the + // Rust node mints a new peer ID per start, and with only the Env + // keyring so does the Go node; a replicator pointed at the old id + // never reconnects. + .with_file_keyring(); + for (i, s) in specs.iter().enumerate() { + builder = builder.with_node_store(i, s.store.as_str()); + } + if a.profile.is_encrypted() { + // Recipe configuration (spec 62, D1): dev mode so Go's KMS has a node + // identity under --no-keyring, an explicit identity per node, and one + // SE key seeded into every file keyring. + builder = builder + .with_encryption() + .with_development() + .with_shared_searchable_encryption_key(SE_KEY); + for (i, spec) in specs.iter().enumerate() { + let bin = spec.kind.host_binary()?; + let bin = &bin; + let identity = generate_identity(bin, &format!("node {i}"))?; + builder = builder.with_node_identity(i, identity.key_hex); + } + println!("encrypted profile: encryption + dev mode + per-node identities + shared SE key"); + } + if a.profile.is_acp() { + builder = builder.with_acp_local(); + } + if a.nac { + builder = builder.with_acp_local().with_nac(); + } + if a.transport == Transport::Iroh { + // Under iroh the builder would otherwise build the workspace with + // the feature; `DEFRA_RUST_BINARY` must already carry it. + let bin = std::env::var("DEFRA_RUST_BINARY").wrap_err( + "--transport iroh needs DEFRA_RUST_BINARY, a `defra` built with --features iroh", + )?; + builder = builder + .with_iroh_transport() + .with_rust_binary(BinarySource::Path(bin.into())); + } + if let Some(intervals) = &a.retry_intervals { + let flag = ["--replicator-retry-intervals", intervals.as_str()]; + builder = builder.with_extra_rust_args(flag).with_extra_go_args(flag); + println!("replicator retry intervals on both nodes: {intervals}s"); + } + let cluster = builder + .build() + .await + .wrap_err("building the mixed cluster")?; + Ok(Nodes::Process { + cluster, + stopped: HashMap::new(), + specs, + }) +} + +/// Everything between the nodes coming up and the final manifest; the +/// caller shuts the nodes down whether or not this succeeds. +async fn drive(run_dir: &Path, a: &RunArgs, nodes: &mut Nodes) -> Result<()> { + for i in 0..nodes.len() { + println!( + "{} at {} ({})", + nodes.name(i), + nodes.api_url(i), + nodes.store(i) + ); + } + let identities = if a.profile.is_acp() { + let rust_node = nodes.first_of(nodes::NodeKind::Rust).ok_or_else(|| { + eyre!("the acp profile mints identities with the Rust CLI, and this topology has no Rust node") + })?; + let bins = nodes.binaries()?; + let rust = &bins[rust_node]; + let ids = Identities { + owner: generate_identity(rust, "owner")?, + reader: generate_identity(rust, "reader")?, + }; + println!( + "acp profile: local ACP, owner {} reader {}", + ids.owner.did, ids.reader.did + ); + Some(ids) + } else { + None + }; + wire_full_mesh(nodes, &a.profile, identities.as_ref(), a.no_subscribe)?; + let preflight_col = if a.profile.relation { + "Author" + } else { + a.profile.collection.as_str() + }; + preflight(nodes, preflight_col, a.profile.relation).await?; + if let Some(ids) = &identities { + token_probe(nodes, &a.profile.collection, &ids.owner).await?; + } + let mut collections = mesh_collections(&a.profile); + if a.control { + wire_control(nodes).await?; + collections.push(CONTROL.to_string()); + } + + let endpoints: Vec<(String, String)> = (0..nodes.len()) + .map(|i| (nodes.name(i).to_string(), nodes.api_url(i))) + .collect(); + let http = http_client(Duration::from_secs(30)); + let mut peer_ids = Vec::new(); + for (name, url) in &endpoints { + let pid = churn::peer_id(&http, url).await; + println!("{name} peer id: {}", pid.as_deref().unwrap_or("?")); + peer_ids.push(pid); + } + // On wall time the run ends at `--secs` if that comes first; events + // drawn past it would never fire. + let mut horizon_ms = churn::virtual_ms(a.ops as u64, a.profile.rate); + if let (Some(cfg), Some(secs)) = (a.churn.as_ref(), a.secs) { + if cfg.clock == ChurnClock::Wall { + horizon_ms = horizon_ms.min(secs * 1000); + } + } + let churn_events = match (&a.churn_schedule, &a.churn) { + (Some(stored), _) => stored.clone(), + (None, Some(cfg)) => churn::schedule_with( + a.seed, + endpoints.len(), + horizon_ms, + cfg, + nodes.supports_partition(), + ), + (None, None) => Vec::new(), + }; + for e in &churn_events { + println!( + "churn plan #{} at {}s: {:?} {} (down {}ms)", + e.index, + e.virtual_ts_ms / 1000, + e.kind, + endpoints[e.node].0, + e.down_ms + ); + } + let run_id = run_dir + .file_name() + .map(|n| n.to_string_lossy().into_owned()) + .unwrap_or_default(); + let rust_binary = std::env::var("DEFRA_RUST_BINARY").unwrap_or_default(); + let backend = if a.docker { "docker" } else { "process" }; + let manifest_path = run_dir.join("manifest.json"); + let manifest = json!({ + "run_id": run_id, + "replay_of": a.replay_of, + "seed": a.seed, + // Null means the backend default, so an older manifest replays as the + // shape it actually ran. + "topology": a.topology.map(Topology::label), + "ops": a.ops, + "secs": a.secs, + "profile": a.profile, + "control": a.control, + "identities": identities, + "nodes": endpoints.iter().zip(&peer_ids).enumerate().map(|(i, ((name, url), pid))| { + let c = nodes.container(i); + json!({ + "name": name, "api_url": url, "store": nodes.store(i), "peer_id": pid, + "backend": backend, "image": c.map(|c| &c.image), + "ip": c.and_then(|c| c.ip.as_deref()), "host": c.map(|c| c.spec.host), + }) + }).collect::>(), + "rust_binary": rust_binary, + "rust_version": version_json(&rust_binary), + "go_compat_commit": std::env::var("DEFRA_GO_COMPAT_COMMIT").unwrap_or_default(), + "go_version": version_json("defradb"), + "churn": a.churn.as_ref().map(|cfg| json!({"config": cfg, "schedule": churn_events})), + "caps": { + "ceiling_bytes": a.ceiling_bytes, "floor_rate": a.floor_rate, + "meter_secs": a.meter_interval.as_secs(), "grace_secs": a.grace.as_secs(), + "settle_secs": a.settle.as_secs(), "min_settle_secs": a.min_settle.as_secs(), + "until_op": a.until_op, + "retry_intervals": a.retry_intervals, "sse_go": a.sse_go, + "no_subscribe": a.no_subscribe, "node_env": a.node_env, + }, + "started_wall_ts_ms": now_ms(), + }); + std::fs::write(&manifest_path, serde_json::to_string_pretty(&manifest)?)?; + + let op_index = Arc::new(AtomicU64::new(0)); + let rate_milli = Arc::new(AtomicU64::new((a.profile.rate * 1000.0) as u64)); + let stop_flag = Arc::new(AtomicBool::new(false)); + let churn_failed = Arc::new(AtomicBool::new(false)); + let workload_done = Arc::new(AtomicBool::new(false)); + let (touched_tx, touched_rx) = mpsc::unbounded_channel(); + let (transitions_tx, transitions_rx) = mpsc::unbounded_channel(); + let (arrivals_tx, arrivals_rx) = mpsc::unbounded_channel(); + let subscription = format!("subscription {{ {} {{ _docID }} }}", a.profile.collection); + let subscriptions: Vec<_> = endpoints + .iter() + .enumerate() + .filter(|(_, (name, _))| a.sse_go || name.starts_with("rust")) + .map(|(i, (_, url))| { + tokio::spawn(sse::subscribe( + i, + url.clone(), + subscription.clone(), + arrivals_tx.clone(), + )) + }) + .collect(); + drop(arrivals_tx); + let (stop_tx, stop_rx) = oneshot::channel(); + let (churn_stop_tx, churn_stop_rx) = oneshot::channel(); + let checker = Checker::new( + endpoints.clone(), + collections, + CheckerConfig { + grace: a.grace, + settle: a.settle, + min_settle: a.min_settle, + encrypted_fields: a.profile.encrypt_fields.clone(), + acp: identities.clone(), + ..CheckerConfig::default() + }, + a.seed, + run_id.clone(), + Arc::clone(&op_index), + run_dir, + )?; + let checker_task = tokio::spawn(checker.run(touched_rx, transitions_rx, arrivals_rx, stop_rx)); + let meter = Meter::new( + MeterConfig { + interval: a.meter_interval, + ceiling_bytes: a.ceiling_bytes, + floor_rate: a.floor_rate, + profile_rate: a.profile.rate, + deadline: a.secs.map(|s| Instant::now() + Duration::from_secs(s)), + ops: a.ops, + }, + run_dir, + Arc::clone(&rate_milli), + Arc::clone(&stop_flag), + Arc::clone(&op_index), + )?; + + let mut generator = Generator::new(a.seed, a.profile.clone(), endpoints.len()); + let mut executor = Executor::new( + endpoints.clone(), + &a.profile, + identities.clone(), + nodes.binaries()?, + &run_dir.join("ops.jsonl"), + )?; + let workload = async { + let result = async { + let mut current_rate = a.profile.rate; + let mut tick = tokio::time::interval(Duration::from_secs_f64(1.0 / current_rate)); + let deadline = a.secs.map(|s| Instant::now() + Duration::from_secs(s)); + let (mut ok, mut failed, mut skipped, mut executed) = (0u64, 0u64, 0u64, 0u64); + let mut stopped_by = "ops"; + let started = Instant::now(); + for op in generator.by_ref().take(a.ops) { + if a.until_op.is_some_and(|n| op.index >= n) { + stopped_by = "until_op"; + break; + } + if deadline.is_some_and(|d| Instant::now() >= d) { + stopped_by = "secs"; + break; + } + if stop_flag.load(Ordering::Relaxed) { + stopped_by = if churn_failed.load(Ordering::Relaxed) { + "churn_error" + } else { + "budget" + }; + break; + } + let rate = rate_milli.load(Ordering::Relaxed) as f64 / 1000.0; + if rate > 0.0 && (rate - current_rate).abs() > 1e-9 { + current_rate = rate; + tick = tokio::time::interval(Duration::from_secs_f64(1.0 / rate)); + tick.tick().await; + } + tick.tick().await; + let record = executor.execute(&op).await?; + op_index.store(op.index + 1, Ordering::Relaxed); + executed = op.index + 1; + if record.ok { + ok += 1; + // Grants change relationships, not documents; nothing replicates. + if let Some(id) = record.doc_id.as_ref().filter(|_| op.kind != OpKind::Grant) { + let _ = touched_tx.send(Touch { + doc_id: id.clone(), + node: record.node.clone(), + wall_ts_ms: record.wall_ts_ms, + create: op.kind == OpKind::Create, + protected: match (op.kind, op.actor) { + (OpKind::Create, Some(Actor::Owner)) => Some(true), + (OpKind::Create, Some(Actor::Anon)) => Some(false), + _ => None, + }, + }); + } + } else if record.skipped { + skipped += 1; + } else { + failed += 1; + println!( + "op {} {:?} on {} failed: {}", + op.index, + op.kind, + record.node, + record.error.unwrap_or_default() + ); + } + } + println!( + "done: {executed} ops ({ok} ok, {failed} failed, {skipped} skipped orphans), {:.2} ops/s over {:.1}s, stopped by {stopped_by}", + executed as f64 / started.elapsed().as_secs_f64(), + started.elapsed().as_secs_f64() + ); + Ok::<_, eyre::Report>((executed, stopped_by)) + } + .await; + // The settle starts here, while the churner (and its meter) keep + // running; an error path must signal it too or the checker never + // returns and the churner is never released. + workload_done.store(true, Ordering::Relaxed); + let _ = stop_tx.send(()); + result + }; + // The churner borrows the nodes from here and shares this task with the + // workload: the harness restart future is not Send, so it cannot be + // spawned. With no schedule it only meters and waits for stop. + let churner = churn::run( + nodes, + churn_events, + a.churn.as_ref().map(|c| c.clock).unwrap_or_default(), + a.profile.rate, + a.profile.collection.clone(), + Arc::clone(&op_index), + transitions_tx, + run_dir.join("topology.jsonl"), + meter, + Arc::clone(&workload_done), + churn_stop_rx, + ); + // After a churner failure the mesh is not what the run planned; stop the + // workload within a tick instead of letting it run for hours against it. + let churner = async { + let result = churner.await; + if result.is_err() { + churn_failed.store(true, Ordering::Relaxed); + stop_flag.store(true, Ordering::Relaxed); + } + result + }; + // The churner holds the nodes, so it is the only thing that can meter: + // release it once the checker's settle and final sweep are done, not when + // the workload stops, or a `--min-settle` run has no idle sample. + let settle_meter = async { + let summary = checker_task.await; + let _ = churn_stop_tx.send(()); + summary + }; + let (workload_result, churner_result, checker_result) = + tokio::join!(workload, churner, settle_meter); + churner_result.wrap_err("churner")?; + let (executed, stopped_by) = workload_result?; + let summary = checker_result?.wrap_err("checker")?; + for task in &subscriptions { + task.abort(); + } + println!( + "subscription events per node: {:?}; checks triggered by quiescence: {}", + summary.sse_events, summary.quiet_checks + ); + println!( + "checks: {} ({} unreachable), divergence records: {}, still present at final sweep: {}", + summary.checks, summary.unreachable, summary.divergences, summary.unresolved + ); + if summary.m5_docs > 0 { + println!("m5 docs compared: {}", summary.m5_docs); + } + if summary.m6_docs + summary.m6_by_design > 0 { + println!( + "m6 docs compared: {} (by design skipped: {})", + summary.m6_docs, summary.m6_by_design + ); + } + println!( + "final sweep: {} mismatches ({})", + summary.final_mismatches, + if summary.final_eligible { + "eligible" + } else { + "NOT eligible: a node was down or in grace" + } + ); + println!( + "divergent docs: {} with a known-cause tag, {} UNTAGGED", + summary.tagged_docs, summary.untagged_docs + ); + if a.hold { + println!("holding: nodes stay up for inspection, press Enter to stop"); + for (name, url) in &endpoints { + println!(" {name}: {url}/api/v0/graphql"); + } + let _ = std::io::stdin().read_line(&mut String::new()); + } + + let mut m: Value = serde_json::from_str(&std::fs::read_to_string(&manifest_path)?)?; + m["ops_executed"] = json!(executed); + m["stopped_by"] = json!(stopped_by); + m["ended_wall_ts_ms"] = json!(now_ms()); + m["checker"] = json!({ + "checks": summary.checks, "unreachable": summary.unreachable, + "divergence_records": summary.divergences, "unresolved": summary.unresolved, + "final_mismatches": summary.final_mismatches, "final_eligible": summary.final_eligible, + "tagged_docs": summary.tagged_docs, "untagged_docs": summary.untagged_docs, + "sse_events": summary.sse_events, "quiet_checks": summary.quiet_checks, + "m5_docs": summary.m5_docs, + "m6_docs": summary.m6_docs, "m6_by_design": summary.m6_by_design, + }); + std::fs::write(&manifest_path, serde_json::to_string_pretty(&m)?)?; + Ok(()) +} + +/// Every node replicates `collection` to every other over libp2p: connect +/// to all peers, subscribe the collection, one replicator per directed +/// pair. Same call order as defradb.rs `p2p_interop_bench`, which is +/// proven on mixed clusters. On ACP the owner adds the policy on every +/// node (ids must agree, the schema references one) and adds the schema. +/// The topics each node subscribes to: none under `--no-subscribe`, which +/// leaves the replicators as the only delivery path. +fn subscribe_topics(collections: &[String], no_subscribe: bool) -> Vec<&str> { + if no_subscribe { + Vec::new() + } else { + collections.iter().map(|c| c.as_str()).collect() + } +} + +fn wire_full_mesh( + nodes: &Nodes, + profile: &Profile, + identities: Option<&Identities>, + no_subscribe: bool, +) -> Result<()> { + let collections = mesh_collections(profile); + let n = nodes.len(); + let addrs: Vec = (0..n).map(|i| nodes.p2p_addr(i)).collect(); + if let Some(ids) = identities { + let mut policy_ids = Vec::new(); + for i in 0..n { + let name = nodes.name(i); + let out = nodes + .client(i) + .acp_policy_add(defra_harness::USER_ACP_POLICY, &ids.owner.key_hex) + .wrap_err_with(|| format!("adding the policy on {name}"))?; + let id = out["PolicyID"] + .as_str() + .or(out["policyID"].as_str()) + .ok_or_else(|| eyre!("policy id missing on {name}: {out}"))?; + policy_ids.push(id.to_string()); + } + eyre::ensure!( + policy_ids.iter().all(|x| x == &policy_ids[0]), + "policy ids differ across nodes: {policy_ids:?}" + ); + println!("policy {} on every node", policy_ids[0]); + for i in 0..n { + nodes + .client(i) + .schema_add_with_identity(&acp_schema(&policy_ids[0]), &ids.owner.key_hex) + .wrap_err_with(|| format!("adding the schema on {}", nodes.name(i)))?; + } + } else { + for i in 0..n { + nodes.client(i).schema_add(schema_for(profile))?; + } + } + for i in 0..n { + let others: Vec<&str> = (0..n) + .filter(|j| *j != i) + .map(|j| addrs[j].as_str()) + .collect(); + nodes.client(i).p2p_connect(&others)?; + } + let topics = subscribe_topics(&collections, no_subscribe); + if topics.is_empty() { + println!("no collection subscribe: replicators are the only delivery path"); + } else { + for i in 0..n { + nodes.client(i).p2p_collection_add(&topics)?; + } + } + let collection_refs: Vec<&str> = collections.iter().map(|c| c.as_str()).collect(); + for i in 0..n { + for j in (0..n).filter(|j| *j != i) { + nodes + .client(i) + .p2p_replicator_set(&collection_refs, &addrs[j])?; + } + } + if let Some(field) = &profile.se_field { + for i in 0..n { + nodes + .client(i) + .encrypted_index_add(&collections[0], field) + .wrap_err_with(|| format!("encrypted index on {}", nodes.name(i)))?; + } + } + Ok(()) +} + +/// T0 check: a doc created on each node must show up on every other node +/// over HTTP GraphQL before any workload runs, so a miswired mesh fails fast. +async fn preflight(nodes: &Nodes, collection: &str, relation: bool) -> Result<()> { + let n = nodes.len(); + for i in 0..n { + let doc = if relation { + format!( + r#"{{"name": "preflight-{}", "age": 1, "verified": true}}"#, + nodes.name(i) + ) + } else { + format!(r#"{{"name": "preflight-{}"}}"#, nodes.name(i)) + }; + nodes + .client(i) + .collection_create(collection, &doc) + .wrap_err_with(|| format!("creating the preflight doc on {}", nodes.name(i)))?; + } + let http = http_client(Duration::from_secs(30)); + let deadline = Instant::now() + Duration::from_secs(90); + loop { + let mut missing = Vec::new(); + for creator in 0..n { + let name = nodes.name(creator); + let query = format!( + "{{ {collection}(filter: {{name: {{_eq: \"preflight-{name}\"}}}}) {{ _docID }} }}" + ); + for viewer in (0..n).filter(|v| *v != creator) { + let data = gql(&http, &nodes.api_url(viewer), &query) + .await + .map_err(eyre::Report::msg)?; + if data[collection].as_array().map_or(0, Vec::len) != 1 { + missing.push(format!("{name} -> {}", nodes.name(viewer))); + } + } + } + if missing.is_empty() { + println!("preflight ok: every node's doc reached every other node"); + return Ok(()); + } + eyre::ensure!( + Instant::now() < deadline, + "preflight docs did not replicate within 90s: {}", + missing.join(", ") + ); + tokio::time::sleep(Duration::from_millis(250)).await; + } +} + +/// A bearer token minted for the owner must be accepted by one node of each +/// runtime (audience = host:port), or every identity-scoped op would fail. +async fn token_probe(nodes: &Nodes, collection: &str, owner: &Identity) -> Result<()> { + let http = http_client(Duration::from_secs(30)); + let query = format!("{{ {collection}(limit: 1) {{ _docID }} }}"); + let probes: Vec = [nodes::NodeKind::Rust, nodes::NodeKind::Go] + .into_iter() + .filter_map(|k| nodes.first_of(k)) + .collect(); + for i in probes.iter().copied() { + let name = nodes.name(i); + let url = nodes.api_url(i); + let token = auth_token(&owner.key_hex, &url)?; + gql_as(&http, &url, &query, Some(&token)) + .await + .map_err(|e| eyre!("bearer token rejected by {name}: {e}"))?; + } + println!( + "token probe ok: owner bearer accepted by {}", + probes + .iter() + .map(|&i| nodes.name(i)) + .collect::>() + .join(" and ") + ); + Ok(()) +} + +/// Positive control: `Control` exists on every node but replicates rust-0 -> +/// go-0 only. A doc created on go-0 never reaches the others (M1), and a +/// rust-0-created doc updated on go-0 has different heads on the two (M3). +async fn wire_control(nodes: &Nodes) -> Result<()> { + for i in 0..nodes.len() { + nodes.client(i).schema_add(CONTROL_SCHEMA)?; + } + let (Some(rust0), Some(go0)) = ( + nodes.first_of(nodes::NodeKind::Rust), + nodes.first_of(nodes::NodeKind::Go), + ) else { + println!( + "skipping --control: it replicates one Rust node to one Go node, and this topology runs a single runtime" + ); + return Ok(()); + }; + let go_addr = nodes.p2p_addr(go0); + nodes + .client(rust0) + .p2p_replicator_set(&[CONTROL], &go_addr)?; + let http = http_client(Duration::from_secs(30)); + let go_url = &nodes.api_url(go0); + let rust_url = &nodes.api_url(rust0); + let create = |v: u32| format!("mutation {{ add_{CONTROL}(input: [{{v: {v}}}]) {{ _docID }} }}"); + gql(&http, go_url, &create(1)) + .await + .map_err(eyre::Report::msg)?; + let data = gql(&http, rust_url, &create(2)) + .await + .map_err(eyre::Report::msg)?; + let id = data[format!("add_{CONTROL}")][0]["_docID"] + .as_str() + .ok_or_else(|| eyre!("control create returned no _docID"))? + .to_string(); + let deadline = Instant::now() + Duration::from_secs(60); + let seen = format!("{{ {CONTROL}(docID: \"{id}\") {{ _docID }} }}"); + loop { + let data = gql(&http, go_url, &seen).await.map_err(eyre::Report::msg)?; + if data[CONTROL].as_array().map_or(0, Vec::len) == 1 { + break; + } + eyre::ensure!(Instant::now() < deadline, "control doc did not reach go-0"); + tokio::time::sleep(Duration::from_millis(250)).await; + } + let update = + format!("mutation {{ update_{CONTROL}(docID: \"{id}\", input: {{v: 3}}) {{ _docID }} }}"); + gql(&http, go_url, &update) + .await + .map_err(eyre::Report::msg)?; + println!("control wired: M1 doc on go-0 only, M3 doc {id} updated on go-0 only"); + Ok(()) +} + +/// ` version --format json`, or null if that fails. +fn version_json(binary: &str) -> Value { + Command::new(binary) + .args(["version", "--format", "json"]) + .output() + .ok() + .and_then(|o| serde_json::from_slice(&o.stdout).ok()) + .unwrap_or(Value::Null) +} + +/// `--name value` from argv. +fn flag(name: &str) -> Option { + let mut args = std::env::args().skip(1); + while let Some(arg) = args.next() { + if arg == format!("--{name}") { + return args.next(); + } + } + None +} + +/// Every `--name VALUE` occurrence, in command-line order. +fn flags(name: &str) -> Vec { + let mut out = Vec::new(); + let mut args = std::env::args().skip(1); + while let Some(arg) = args.next() { + if arg == format!("--{name}") { + out.extend(args.next()); + } + } + out +} + +/// The only validation a `--node-env` item gets: it has an `=`. +fn split_node_env(item: &str) -> Result<(&str, &str)> { + item.split_once('=') + .ok_or_else(|| eyre!("--node-env {item}: expected KEY=VALUE")) +} + +/// Set the pairs on this process; spawned nodes inherit them. +fn apply_node_env(items: &[String]) -> Result<()> { + for item in items { + let (k, v) = split_node_env(item)?; + std::env::set_var(k, v); + } + if !items.is_empty() { + println!("node env on every node: {}", items.join(" ")); + } + Ok(()) +} + +fn has_flag(name: &str) -> bool { + std::env::args().any(|a| a == format!("--{name}")) +} + +fn parse_flag(name: &str, default: T) -> Result +where + T: std::str::FromStr, + T::Err: std::fmt::Display, +{ + Ok(opt_flag(name)?.unwrap_or(default)) +} + +fn opt_flag(name: &str) -> Result> +where + T: std::str::FromStr, + T::Err: std::fmt::Display, +{ + flag(name) + .map(|s| s.parse::()) + .transpose() + .map_err(|e| eyre!("--{name}: {e}")) +} + +fn unix_secs() -> u64 { + SystemTime::now() + .duration_since(UNIX_EPOCH) + .map_or(0, |d| d.as_secs()) +} + +/// `runs/-/` under the current dir, absolute. Fails if it +/// exists so two runs never share an artifact. +fn new_run_dir(seed: u64) -> Result { + std::fs::create_dir_all("runs")?; + let dir = PathBuf::from("runs").join(format!("{}-{seed}", unix_secs())); + std::fs::create_dir(&dir).wrap_err_with(|| format!("creating {}", dir.display()))?; + Ok(dir.canonicalize()?) +} + +#[cfg(test)] +mod tests { + use super::*; + + /// `--no-subscribe` removes gossip so the replicator is the only path; + /// without the flag every node subscribes to the collection as before. + #[test] + fn no_subscribe_skips_the_collection_topic() { + let users = vec!["Users".to_string()]; + assert_eq!(subscribe_topics(&users, false), vec!["Users"]); + assert!(subscribe_topics(&users, true).is_empty()); + let both = vec!["Author".to_string(), "Book".to_string()]; + assert_eq!(subscribe_topics(&both, false), vec!["Author", "Book"]); + } + + /// The process backend has no per-node env hook: nodes are children of + /// this process and inherit its environment, so applying the pairs here + /// is what reaches both runtimes. + #[test] + fn node_env_is_applied_to_this_process_and_validated() { + assert_eq!( + split_node_env("RUST_LOG=debug").unwrap(), + ("RUST_LOG", "debug") + ); + // A value may contain further `=`; only the first splits. + assert_eq!( + split_node_env("RUST_LOG=defra=debug,libp2p=info").unwrap(), + ("RUST_LOG", "defra=debug,libp2p=info") + ); + assert!(split_node_env("RUST_LOG").is_err()); + + apply_node_env(&["SOAK_TEST_NODE_ENV=debug".to_string()]).unwrap(); + assert_eq!(std::env::var("SOAK_TEST_NODE_ENV").unwrap(), "debug"); + assert!(apply_node_env(&["NO_EQUALS_HERE".to_string()]).is_err()); + // Nothing to apply, nothing set. + apply_node_env(&[]).unwrap(); + assert!(std::env::var("SOAK_TEST_NODE_ENV_UNSET").is_err()); + } + + /// The manifest cap is `caps.node_env`, and a replay of a run that used + /// the flag gets the same environment without retyping it. + #[test] + fn node_env_round_trips_through_manifest_caps() { + let dir = std::env::temp_dir().join(format!("soak-caps-{}", std::process::id())); + std::fs::create_dir_all(&dir).unwrap(); + let path = dir.join("manifest.json"); + std::fs::write( + &path, + serde_json::to_string(&json!({ + "run_id": "1789000000-42", + "seed": 42, + "ops": 100, + "profile": Profile::p0_crud(), + "nodes": [{"backend": "process"}], + "caps": {"node_env": ["RUST_LOG=debug", "GOLOG_LEVEL=debug"]}, + })) + .unwrap(), + ) + .unwrap(); + let a = RunArgs::from_manifest(&path).unwrap(); + assert_eq!(a.node_env, ["RUST_LOG=debug", "GOLOG_LEVEL=debug"]); + // A manifest written before this commit has no such cap. + std::fs::write( + &path, + serde_json::to_string(&json!({ + "run_id": "1789000000-42", "seed": 42, "ops": 100, + "profile": Profile::p0_crud(), "nodes": [{"backend": "process"}], "caps": {}, + })) + .unwrap(), + ) + .unwrap(); + assert!(RunArgs::from_manifest(&path).unwrap().node_env.is_empty()); + std::fs::remove_dir_all(&dir).ok(); + } + + #[test] + fn topology_parses_and_rejects_empty_meshes() { + assert_eq!( + Topology::parse("2r2g").unwrap(), + Topology { rust: 2, go: 2 } + ); + assert_eq!( + Topology::parse("4r0g").unwrap(), + Topology { rust: 4, go: 0 } + ); + assert_eq!( + Topology::parse("0r4g").unwrap(), + Topology { rust: 0, go: 4 } + ); + assert_eq!(Topology::parse("10r3g").unwrap().total(), 13); + for bad in ["0r0g", "2r2", "r2g", "2g2r", "", "2r2gx", "-1r2g"] { + assert!(Topology::parse(bad).is_err(), "{bad} should be rejected"); + } + } + + #[test] + fn transport_parses_and_defaults_to_libp2p() { + assert_eq!(Transport::parse(None).unwrap(), Transport::Libp2p); + assert_eq!(Transport::parse(Some("libp2p")).unwrap(), Transport::Libp2p); + assert_eq!(Transport::parse(Some("iroh")).unwrap(), Transport::Iroh); + assert!(Transport::parse(Some("quic")).is_err()); + assert_eq!(Transport::Iroh.label(), "iroh"); + } + + #[test] + fn topology_round_trips_through_the_manifest_label() { + for spec in ["2r2g", "4r0g", "0r4g", "6r2g"] { + let t = Topology::parse(spec).unwrap(); + assert_eq!(t.label(), spec); + assert_eq!(Topology::parse(&t.label()).unwrap(), t); + } + } + + /// An older manifest has no `topology`, and must replay as the shape it + /// actually ran: the M2 six in containers, two of each as processes. + #[test] + fn absent_topology_keeps_each_backend_default() { + assert_eq!(node_specs(None, true).len(), 6); + assert_eq!(node_specs(None, false).len(), 4); + assert_eq!(node_specs(Some(Topology { rust: 0, go: 3 }), true).len(), 3); + } + + #[test] + fn default_process_topology_matches_the_published_shape() { + // Every published process run is two regolith nodes then two badger + // ones; the default must keep naming and order identical. + let s = nodes::topology_specs(2, 2); + assert_eq!( + s.iter().map(|n| n.store.as_str()).collect::>(), + ["regolith", "regolith", "badger", "badger"] + ); + } + + #[test] + fn schema_follows_profile() { + assert_eq!(schema_for(&Profile::p0_crud()), SCHEMA); + assert_eq!(schema_for(&Profile::p1_encrypted()), VAULT_SCHEMA); + // The ACP path uses `acp_schema` instead; pin what `schema_for` returns. + assert_eq!(schema_for(&Profile::p2_acp()), SCHEMA); + assert_eq!(schema_for(&Profile::p0_index()), INDEXED_SCHEMA); + assert_eq!(schema_for(&Profile::p3_relation()), RELATION_SCHEMA); + assert!(INDEXED_SCHEMA.contains("@index")); + assert!(!SCHEMA.contains("@index")); + assert!(RELATION_SCHEMA.contains("type Book") && RELATION_SCHEMA.contains("type Author")); + assert!(!RELATION_SCHEMA.contains("@index")); + assert!(VAULT_SCHEMA.contains("type Vault") && VAULT_SCHEMA.contains("secret: String")); + } + + #[test] + fn acp_schema_carries_policy_id() { + assert_eq!( + acp_schema("abc"), + "type User @policy(id: \"abc\", resource: \"users\") { name: String age: Int score: Float blob: String }" + ); + } +} diff --git a/crates/soak/src/manage/actors.rs b/crates/soak/src/manage/actors.rs new file mode 100644 index 0000000..0829276 --- /dev/null +++ b/crates/soak/src/manage/actors.rs @@ -0,0 +1,81 @@ +//! The three identities the cases speak as, and their NAC grants on every +//! target. Grants go through `acp node relationship add` as the cluster's +//! startup identity (the NAC owner), live, no restart. + +use std::cell::RefCell; +use std::collections::HashMap; +use std::path::Path; + +use eyre::{Result, WrapErr}; +use serde::Serialize; + +use crate::auth::{manage_token, Identity}; +use crate::nodes::Nodes; + +#[derive(Clone, Copy, Debug, PartialEq, Eq, Hash, Serialize)] +pub enum Actor { + /// The `admin` relation: every node permission. + Admin, + /// `add-p2p-collection` and `list-p2p-replicator` only, so `CollectionAdd` + /// lands and `ReplicatorAdd` is refused from the same actor. Granted on + /// every run. + Operator, + /// No grants. + Outsider, +} + +pub const OPERATOR_GRANTS: &[&str] = &["add-p2p-collection", "list-p2p-replicator"]; + +#[derive(Serialize)] +pub struct Actors { + pub admin: Identity, + pub operator: Identity, + pub outsider: Identity, + /// Manage tokens per (actor, target peer id); they outlive a run. + #[serde(skip)] + tokens: RefCell>, +} + +impl Actors { + pub fn generate(bin: &Path) -> Result { + Ok(Self { + admin: crate::generate_identity(bin, "admin")?, + operator: crate::generate_identity(bin, "operator")?, + outsider: crate::generate_identity(bin, "outsider")?, + tokens: RefCell::default(), + }) + } + + /// Apply the grants on node `i` as `owner_key`. + pub fn grant_on(&self, nodes: &Nodes, i: usize, owner_key: &str) -> Result<()> { + let client = nodes.client(i); + let name = nodes.name(i); + client + .acp_node_relationship_add("admin", &self.admin.did, owner_key) + .wrap_err_with(|| format!("granting admin on {name}"))?; + for relation in OPERATOR_GRANTS { + client + .acp_node_relationship_add(relation, &self.operator.did, owner_key) + .wrap_err_with(|| format!("granting {relation} on {name}"))?; + } + Ok(()) + } + + pub fn identity(&self, actor: Actor) -> &Identity { + match actor { + Actor::Admin => &self.admin, + Actor::Operator => &self.operator, + Actor::Outsider => &self.outsider, + } + } + + pub fn token(&self, actor: Actor, target_peer_id: &str) -> Result { + let k = (actor, target_peer_id.to_string()); + if let Some(tok) = self.tokens.borrow().get(&k) { + return Ok(tok.clone()); + } + let tok = manage_token(&self.identity(actor).key_hex, target_peer_id)?; + self.tokens.borrow_mut().insert(k, tok.clone()); + Ok(tok) + } +} diff --git a/crates/soak/src/manage/authz.rs b/crates/soak/src/manage/authz.rs new file mode 100644 index 0000000..7539113 --- /dev/null +++ b/crates/soak/src/manage/authz.rs @@ -0,0 +1,569 @@ +//! Authorization cases: who the target lets do what, and when. + +use eyre::Result; +use serde_json::json; + +use super::actors::{Actor, OPERATOR_GRANTS}; +use super::cases::{ + collection_add, collection_remove, every_op, expect_status, fail, managed_state, Channel, Verb, + COLLECTION, +}; + +/// The NAC relation an op's permission is (`ManageMutateOp::permission`, +/// `ManageQueryOp::permission` and `NodePermission::as_str` in defradb.rs). +fn permission(kind: &str) -> &'static str { + match kind { + "ReplicatorAdd" => "add-p2p-replicator", + "ReplicatorDelete" => "delete-p2p-replicator", + "ReplicatorList" => "list-p2p-replicator", + "CollectionAdd" => "add-p2p-collection", + "CollectionRemove" => "delete-p2p-collection", + "CollectionList" => "list-p2p-collection", + "DocumentAdd" => "add-p2p-document", + "DocumentRemove" => "delete-p2p-document", + "DocumentList" => "list-p2p-document", + "PeerConnect" => "connect-p2p-peer", + "PeerDisconnect" => "disconnect-p2p-peer", + other => unreachable!("{other} is not a manage op"), + } +} + +/// A1: `operator` tries every op; the ones its grants cover land, the rest +/// are refused. One row per op, every miss named with the grant it had. +pub(super) async fn a1(ch: &mut dyn Channel) -> Result<()> { + let (relay, target) = (0, 1); + let mut misses = Vec::new(); + let mut subscribed = false; + for op in every_op(&ch.addr(relay)) { + let kind = op["Kind"].as_str().unwrap_or_default().to_string(); + let relation = permission(&kind); + let granted = OPERATOR_GRANTS.contains(&relation); + let r = ch.send(relay, target, Actor::Operator, op).await?; + if r.status != if granted { 200 } else { 403 } { + misses.push(format!( + "{kind}: {} {} with {relation} {}", + r.status, + r.body, + if granted { "granted" } else { "not granted" } + )); + } + subscribed |= kind == "CollectionAdd" && r.status == 200; + } + if subscribed { + let r = ch + .send(relay, target, Actor::Admin, collection_remove()) + .await?; + expect_status(&r, 200, "CollectionRemove (restore)")?; + } + if misses.is_empty() { + return Ok(()); + } + Err(fail( + format!( + "200 on the ops {} cover, 403 on the rest", + OPERATOR_GRANTS.join(" and ") + ), + misses.join("; "), + )) +} + +/// A2: the outsider is refused on every op with 403 at the relay, and the +/// target's three lists are the same before and after. +pub(super) async fn a2(ch: &mut dyn Channel) -> Result<()> { + let (relay, target) = (0, 1); + let before = managed_state(ch, relay, target).await?; + for op in every_op(&ch.addr(relay)) { + let kind = op["Kind"].as_str().unwrap_or_default().to_string(); + let r = ch.send(relay, target, Actor::Outsider, op).await?; + expect_status(&r, 403, &format!("outsider {kind}"))?; + } + let after = managed_state(ch, relay, target).await?; + if before != after { + return Err(fail( + format!("target state unchanged: {before}"), + after.to_string(), + )); + } + Ok(()) +} + +/// A3: `operator` lands `CollectionAdd`, its grant is revoked on the +/// target, the same op is refused. The post-revoke op goes out even when +/// the first was refused, so the verdict says whether the revoke was +/// enforced, not enforced, or untested because the grant was inert. The +/// grant comes back either way. +pub(super) async fn a3(ch: &mut dyn Channel) -> Result<()> { + let (relay, target) = (0, 1); + let relation = permission("CollectionAdd"); + let actor = Actor::Operator; + let node = target; + let first = ch.send(relay, target, actor, collection_add()).await?; + ch.control(Verb::Revoke { + node, + actor, + relation, + }) + .await?; + let second = ch.send(relay, target, actor, collection_add()).await; + ch.control(Verb::Grant { + node, + actor, + relation, + }) + .await?; + let second = second?; + if first.status == 200 || second.status == 200 { + let r = ch + .send(relay, target, Actor::Admin, collection_remove()) + .await?; + expect_status(&r, 200, "CollectionRemove (restore)")?; + } + match (first.status, second.status) { + (200, 403) => { + ch.note("revoke enforced: 200 before, 403 after".into()); + Ok(()) + } + (200, after) => Err(fail( + format!("403 on operator CollectionAdd after revoking {relation}"), + format!("{after} {}: revoke not enforced", second.body), + )), + (before, after) => Err(fail( + format!("200 on operator CollectionAdd with {relation} granted, before the revoke"), + format!( + "{before} {} before, {after} after; revoke untested: pre-revoke op refused (grant inert, defect 1)", + first.body + ), + )), + } +} + +/// A4: `outsider` is refused `CollectionAdd`, is granted its permission on +/// the target, the same op lands. The grant is revoked either way. +pub(super) async fn a4(ch: &mut dyn Channel) -> Result<()> { + let (relay, target) = (0, 1); + let relation = permission("CollectionAdd"); + let first = ch + .send(relay, target, Actor::Outsider, collection_add()) + .await?; + expect_status(&first, 403, "outsider CollectionAdd before the grant")?; + let actor = Actor::Outsider; + let node = target; + ch.control(Verb::Grant { + node, + actor, + relation, + }) + .await?; + let second = ch + .send(relay, target, Actor::Outsider, collection_add()) + .await; + ch.control(Verb::Revoke { + node, + actor, + relation, + }) + .await?; + let second = second?; + if second.status == 200 { + let r = ch + .send(relay, target, Actor::Admin, collection_remove()) + .await?; + expect_status(&r, 200, "CollectionRemove (restore)")?; + } + expect_status( + &second, + 200, + &format!("outsider CollectionAdd with {relation} granted"), + ) +} + +/// A5: a token minted for node 1 is relayed by node 0 to node 2: refused +/// on audience (400, the token is malformed for node 2) and nothing lands. +pub(super) async fn a5(ch: &mut dyn Channel) -> Result<()> { + let (relay, minted_for, target) = (0, 1, 2); + let r = ch + .send_for(relay, target, minted_for, Actor::Admin, collection_add()) + .await?; + if r.status == 200 { + let rm = ch + .send(relay, target, Actor::Admin, collection_remove()) + .await?; + expect_status(&rm, 200, "CollectionRemove (restore)")?; + } + if r.status != 400 || !r.body.to_string().contains("token") { + return Err(fail( + "400 with a token error on CollectionAdd carrying node 1's audience to node 2", + format!("{} {}", r.status, r.body), + )); + } + let list = ch + .send( + relay, + target, + Actor::Admin, + json!({ "Kind": "CollectionList" }), + ) + .await?; + expect_status(&list, 200, "CollectionList as admin")?; + if list.body["values"] + .as_array() + .is_some_and(|v| v.iter().any(|c| c == COLLECTION)) + { + return Err(fail( + format!("{COLLECTION} absent from node 2's CollectionList after the refused add"), + list.body.to_string(), + )); + } + Ok(()) +} + +/// A6: with NAC disabled on the target (its own `acp node disable`, as the +/// owner) the outsider's ops land; NAC comes back either way. +pub(super) async fn a6(ch: &mut dyn Channel) -> Result<()> { + let (relay, target) = (0, 1); + ch.control(Verb::Nac { + node: target, + on: false, + }) + .await?; + let ops = async { + let r = ch + .send(relay, target, Actor::Outsider, collection_add()) + .await?; + expect_status( + &r, + 200, + "outsider CollectionAdd with NAC disabled on the target", + )?; + let r = ch + .send(relay, target, Actor::Outsider, collection_remove()) + .await?; + expect_status( + &r, + 200, + "outsider CollectionRemove with NAC disabled on the target", + ) + } + .await; + ch.control(Verb::Nac { + node: target, + on: true, + }) + .await?; + ops +} + +#[cfg(test)] +mod tests { + use std::cell::RefCell; + use std::rc::Rc; + + use serde_json::Value; + + use super::super::cases::fake::*; + use super::super::cases::Outcome; + use super::super::client::Reply; + use super::*; + + const ADD: &str = "add-p2p-collection"; + + /// The target honours per-permission grants: `granted` decides the + /// operator's ops, the outsider is always refused. + fn by_grants( + granted: &'static [&'static str], + ) -> impl FnMut(usize, usize, usize, Actor, &Value) -> Result { + move |_, _, _, actor, op| { + let kind = op["Kind"].as_str().unwrap(); + match actor { + Actor::Admin => admin_view(op), + Actor::Operator if granted.contains(&permission(kind)) => admin_view(op), + _ => status(403), + } + } + } + + #[tokio::test] + async fn a1_sends_every_op_as_operator_and_wants_the_grants_honoured() { + let (tx, rx) = std::sync::mpsc::channel(); + let mut grants = by_grants(OPERATOR_GRANTS); + let outcome = run_one("A1", move |relay, target, aud, actor, op| { + if actor == Actor::Operator { + tx.send(op["Kind"].as_str().unwrap().to_string()).unwrap(); + } + if actor == Actor::Admin && op["Kind"] == "CollectionRemove" { + tx.send("restore".into()).unwrap(); + } + grants(relay, target, aud, actor, op) + }) + .await; + assert_eq!(outcome, Outcome::Pass); + let rows: Vec = rx.try_iter().collect(); + assert_eq!(rows.len(), 12, "{rows:?}"); + assert_eq!(rows.iter().filter(|r| *r == "restore").count(), 1); + assert_eq!(rows.iter().filter(|r| *r == "CollectionAdd").count(), 1); + } + + #[tokio::test] + async fn a1_names_the_granted_permission_when_a_granted_op_is_refused() { + let defect = run_one("A1", by_grants(&[])).await; + let Outcome::Fail { expected, got } = &defect else { + panic!("{defect:?}"); + }; + assert!( + expected.contains(ADD) && expected.contains("403"), + "{expected}" + ); + assert!( + got.contains("CollectionAdd: 403") && got.contains(&format!("{ADD} granted")), + "{got}" + ); + assert!(got.contains("ReplicatorList: 403"), "{got}"); + assert!(!got.contains("ReplicatorAdd"), "{got}"); + + let leak = run_one("A1", |_, _, _, _, op| admin_view(op)).await; + assert!( + matches!(&leak, Outcome::Fail { got, .. } if got.contains("ReplicatorAdd: 200") && got.contains("add-p2p-replicator not granted")), + "{leak:?}" + ); + } + + /// A target that consults the live grants: the last verb decides. With + /// `defect`, per-permission grants have no effect (the product today). + fn live_grants(verbs: Rc>>, defect: bool) -> Fake { + let seen = verbs.clone(); + let mut fake = Fake::new(move |_, _, _, actor, op| { + if actor == Actor::Admin { + return admin_view(op); + } + let last = seen.borrow().last().cloned(); + let granted = match (actor, last) { + (Actor::Operator, Some(Verb::Revoke { .. })) => false, + (Actor::Operator, _) => true, + (Actor::Outsider, Some(Verb::Grant { .. })) => true, + (Actor::Outsider, Some(Verb::Nac { on: false, .. })) => return admin_view(op), + _ => false, + }; + if granted && !defect { + admin_view(op) + } else { + status(403) + } + }); + fake.verbs = verbs; + fake + } + + #[tokio::test] + async fn a3_revokes_then_wants_a_403_and_restores_the_grant() { + let verbs: Rc>> = Rc::default(); + let (outcome, notes) = run_noted("A3", live_grants(verbs.clone(), false)).await; + assert_eq!(outcome, Outcome::Pass); + assert_eq!(notes, ["revoke enforced: 200 before, 403 after"]); + let revoke = Verb::Revoke { + node: 1, + actor: Actor::Operator, + relation: ADD, + }; + let grant = Verb::Grant { + node: 1, + actor: Actor::Operator, + relation: ADD, + }; + assert_eq!(*verbs.borrow(), [revoke.clone(), grant.clone()]); + + let (defect, verbs) = run_fake("A3", live_grants(Rc::default(), true)).await; + assert!( + matches!(&defect, Outcome::Fail { expected, got } if expected.contains(ADD) && expected.contains("before") && got.starts_with("403") && got.ends_with("revoke untested: pre-revoke op refused (grant inert, defect 1)")), + "{defect:?}" + ); + assert_eq!(verbs, [revoke, grant]); + + let (stale, verbs) = run_fake( + "A3", + Fake::new(|_, _, _, actor, op| { + if actor == Actor::Admin { + admin_view(op) + } else { + status(200) + } + }), + ) + .await; + assert!( + matches!(&stale, Outcome::Fail { expected, got } if expected.contains("403") && expected.contains("revok") && got.ends_with("revoke not enforced")), + "{stale:?}" + ); + assert_eq!(verbs.len(), 2, "{verbs:?}"); + } + + #[tokio::test] + async fn a4_grants_then_wants_a_200_and_revokes() { + let (outcome, verbs) = run_fake("A4", live_grants(Rc::default(), false)).await; + assert_eq!(outcome, Outcome::Pass); + let grant = Verb::Grant { + node: 1, + actor: Actor::Outsider, + relation: ADD, + }; + let revoke = Verb::Revoke { + node: 1, + actor: Actor::Outsider, + relation: ADD, + }; + assert_eq!(verbs, [grant.clone(), revoke.clone()]); + + let (defect, verbs) = run_fake("A4", live_grants(Rc::default(), true)).await; + assert!( + matches!(&defect, Outcome::Fail { expected, got } if expected.contains(&format!("{ADD} granted")) && got.starts_with("403")), + "{defect:?}" + ); + assert_eq!(verbs, [grant, revoke]); + } + + #[tokio::test] + async fn a5_relays_node_1s_token_to_node_2_and_wants_a_token_error() { + let (tx, rx) = std::sync::mpsc::channel(); + let outcome = run_one("A5", move |relay, target, aud, actor, op| { + tx.send(( + relay, + target, + aud, + actor, + op["Kind"].as_str().unwrap().to_string(), + )) + .unwrap(); + if aud != target { + return Ok(Reply { + status: 400, + body: json!({"error": "transport: actor token rejected: audience"}), + latency_ms: 1, + }); + } + if op["Kind"] == "CollectionList" { + return ok(json!({"Kind": "Strings", "values": []})); + } + admin_view(op) + }) + .await; + assert_eq!(outcome, Outcome::Pass); + let sent: Vec<_> = rx + .try_iter() + .skip_while(|s| s.4 == "CollectionList") + .collect(); + assert_eq!(sent[0], (0, 2, 1, Actor::Admin, "CollectionAdd".into())); + assert!(sent.iter().all(|s| s.0 == 0 && s.1 == 2), "{sent:?}"); + + let leak = run_one("A5", |_, _, _, _, op| { + if op["Kind"] == "CollectionList" { + ok(json!({"Kind": "Strings", "values": [COLLECTION]})) + } else { + admin_view(op) + } + }) + .await; + assert!( + matches!(&leak, Outcome::Fail { expected, got } if expected.contains("400") && got.starts_with("200")), + "{leak:?}" + ); + } + + #[tokio::test] + async fn a6_disables_nac_on_the_target_wants_outsider_ops_to_land_and_re_enables() { + let (outcome, verbs) = run_fake("A6", live_grants(Rc::default(), false)).await; + assert_eq!(outcome, Outcome::Pass); + let off = Verb::Nac { node: 1, on: false }; + let on = Verb::Nac { node: 1, on: true }; + assert_eq!(verbs, [off.clone(), on.clone()]); + + let (refused, verbs) = run_fake( + "A6", + Fake::new(|_, _, _, actor, op| { + if actor == Actor::Admin { + admin_view(op) + } else { + status(403) + } + }), + ) + .await; + assert!( + matches!(&refused, Outcome::Fail { expected, .. } if expected.contains("NAC disabled")), + "{refused:?}" + ); + assert_eq!(verbs, [off, on]); + } + + #[tokio::test] + async fn a2_needs_403_on_every_op_and_unchanged_state() { + let mut kinds = Vec::new(); + let (tx, rx) = std::sync::mpsc::channel(); + let outcome = run_one("A2", move |_, _, _, actor, op| { + if actor == Actor::Outsider { + tx.send(op["Kind"].as_str().unwrap().to_string()).unwrap(); + status(403) + } else { + admin_view(op) + } + }) + .await; + assert_eq!(outcome, Outcome::Pass); + kinds.extend(rx.try_iter()); + kinds.sort(); + assert_eq!(kinds.len(), 11, "{kinds:?}"); + assert!( + kinds.contains(&"PeerDisconnect".to_string()) + && kinds.contains(&"DocumentList".to_string()) + ); + + let leaked = run_one("A2", |_, _, _, actor, op| { + if actor == Actor::Outsider && op["Kind"] == "CollectionAdd" { + status(200) + } else if actor == Actor::Outsider { + status(403) + } else { + admin_view(op) + } + }) + .await; + assert!( + matches!(&leaked, Outcome::Fail { expected, got } if expected.contains("CollectionAdd") && got.starts_with("200")), + "{leaked:?}" + ); + + let mut visited = false; + let drifted = run_one("A2", move |_, _, _, actor, op| { + if actor == Actor::Outsider { + visited = true; + return status(403); + } + if op["Kind"] == "CollectionList" && visited { + return ok(json!({"Kind": "Strings", "values": []})); + } + admin_view(op) + }) + .await; + assert!( + matches!(&drifted, Outcome::Fail { expected, .. } if expected.contains("unchanged")), + "{drifted:?}" + ); + + let mut visited = false; + let refiltered = run_one("A2", move |_, _, _, actor, op| { + if actor == Actor::Outsider { + visited = true; + return status(403); + } + if op["Kind"] == "ReplicatorList" && visited { + return ok(json!({"Kind": "Replicators", "replicators": [ + {"id": "peer0", "address": "/ip4/127.0.0.1/tcp/0/p2p/peer0", "collections": ["bafy-user"], "filters": {"bafy-user": {"Field": "age", "Value": 1}}} + ]})); + } + admin_view(op) + }) + .await; + assert!( + matches!(&refiltered, Outcome::Fail { expected, got } if expected.contains("unchanged") && got.contains("age")), + "{refiltered:?}" + ); + } +} diff --git a/crates/soak/src/manage/bounds.rs b/crates/soak/src/manage/bounds.rs new file mode 100644 index 0000000..e05af66 --- /dev/null +++ b/crates/soak/src/manage/bounds.rs @@ -0,0 +1,561 @@ +//! Bounds cases: the request size the channel carries, and a target that +//! never answers. Big requests are `DocumentAdd` with many well-formed +//! doc ids, the only op whose payload scales. + +use eyre::Result; +use serde_json::{json, Value}; + +use super::actors::Actor; +use super::cases::{expect_status, fail, Channel, Verb, COLLECTION}; +use crate::Transport; + +/// Wire bytes of one doc ref: a CBOR map of `Collection` "User" and a +/// 40-character `DocID`. A request of `n` refs is `n * DOC_REF_BYTES` +/// plus an 855-byte envelope (token, signature), well under `MARGIN`. +const DOC_REF_BYTES: usize = 65; + +/// What the bounds cases size against, per transport. +pub struct Bounds { + /// The request size bound, in wire bytes. + pub max_request: usize, + /// B4's request: big enough that the target is still applying it + /// `pause_after_ms` in, small enough that the target's document list + /// still fits a reply. + pub silent_request: usize, + pub pause_after_ms: u64, +} + +/// libp2p is what B3 located on 2026-09-17: 258048 refs (16773977 B) land, +/// 258112 (16778137 B) are refused with "failed to write manage request: +/// connection is closed" in 2 s, the target's `max_msg_size`. B3 measured +/// 12.5 MiB at 3.2 s round trip, so B4 pauses 1.5 s into 12 MiB. +pub const LIBP2P: Bounds = Bounds { + max_request: 16 * 1024 * 1024, + silent_request: 12 * 1024 * 1024, + pause_after_ms: 1_500, +}; + +/// Iroh is `MAX_MANAGE_MSG_SIZE` (crates/p2p/src/iroh/protocols.rs:98), +/// which B2 confirmed from above on 2026-09-18: 65536 refs are refused in +/// 212 ms with "codec error: failed to write payload: sending stopped by +/// peer: error 0". B3 could not bracket it from below: a `DocumentAdd` of +/// 1024 refs (~65 KiB) gets "response timeout" after 30 s, the target +/// stops accepting after its 165th document-topic join, and every later +/// dial of it is "dial error: timed out". B4's request wedges the target +/// the same way, so its pause has not been calibrated. +pub const IROH: Bounds = Bounds { + max_request: 4 * 1024 * 1024, + silent_request: 3 * 1024 * 1024, + pause_after_ms: 400, +}; + +const _: () = assert!(LIBP2P.silent_request < LIBP2P.max_request); +const _: () = assert!(IROH.silent_request < IROH.max_request); + +pub fn for_transport(t: Transport) -> &'static Bounds { + match t { + Transport::Libp2p => &LIBP2P, + Transport::Iroh => &IROH, + } +} + +/// How far under and over the bound B1 and B2 sit. +const MARGIN: usize = 64 * 1024; + +/// B3 stops when the landed and refused sizes are this close, in refs. +const RESOLUTION: usize = 64; + +/// B3 gives up looking for a refusal past this many refs. +const CAP: usize = 1 << 20; + +fn docs_for(bytes: usize) -> usize { + bytes / DOC_REF_BYTES +} + +fn document_op(kind: &str, n: usize) -> Value { + let docs: Vec = (0..n) + .map(|i| json!({ "collection": COLLECTION, "doc_id": format!("bae-00000000-0000-0000-0000-{i:012x}") })) + .collect(); + json!({ "Kind": kind, "docs": docs }) +} + +/// B1: a `DocumentAdd` just under the bound lands; the same refs are +/// removed again. +pub(super) async fn b1(ch: &mut dyn Channel) -> Result<()> { + let (relay, target) = (0, 1); + let n = docs_for(for_transport(ch.transport()).max_request - MARGIN); + let r = ch + .send(relay, target, Actor::Admin, document_op("DocumentAdd", n)) + .await?; + expect_status(&r, 200, &format!("DocumentAdd of {n} refs under the bound"))?; + let r = ch + .send( + relay, + target, + Actor::Admin, + document_op("DocumentRemove", n), + ) + .await?; + expect_status(&r, 200, "DocumentRemove (restore)") +} + +/// B2: a `DocumentAdd` just over the bound is refused, not served and not +/// a 5xx; the status is recorded. The target then serves a request and +/// so does the relay, for a third node. +pub(super) async fn b2(ch: &mut dyn Channel) -> Result<()> { + let (relay, target, other) = (0, 1, 2); + let n = docs_for(for_transport(ch.transport()).max_request + MARGIN); + let r = ch + .send(relay, target, Actor::Admin, document_op("DocumentAdd", n)) + .await?; + if r.status == 200 || r.status >= 500 { + return Err(fail( + format!("a clean refusal of {n} refs over the bound"), + format!("{} after {} ms: {}", r.status, r.latency_ms, r.body), + )); + } + ch.note(format!( + "{n} refs over the bound: {} after {} ms: {}", + r.status, r.latency_ms, r.body + )); + let list = json!({ "Kind": "CollectionList" }); + let r = ch.send(relay, target, Actor::Admin, list.clone()).await?; + expect_status( + &r, + 200, + "CollectionList on the target after the oversized request", + )?; + let r = ch.send(relay, other, Actor::Admin, list).await?; + expect_status(&r, 200, "CollectionList through the relay to a third node") +} + +/// B3: the transport's bound, by doubling then bisecting the ref count of a +/// `DocumentAdd`; each probe and the located interval are recorded, not +/// asserted. Landed probes are removed again; a refused probe leaves the +/// target as it was, so this runs alone under `--locate-size-bound`. +pub(super) async fn b3(ch: &mut dyn Channel) -> Result<()> { + let (relay, target) = (0, 1); + let mut lo = 0; + let mut hi: Option<(usize, String)> = None; + let mut n = 1024; + while n <= CAP { + let r = ch + .send(relay, target, Actor::Admin, document_op("DocumentAdd", n)) + .await?; + ch.note(format!( + "{n} refs (~{} KiB): {} after {} ms {}", + n * DOC_REF_BYTES / 1024, + r.status, + r.latency_ms, + if r.status == 200 { + String::new() + } else { + r.body.to_string() + } + )); + if r.status == 200 { + lo = n; + let r = ch + .send( + relay, + target, + Actor::Admin, + document_op("DocumentRemove", n), + ) + .await?; + expect_status(&r, 200, &format!("DocumentRemove of {n} refs (restore)"))?; + } else { + hi = Some((n, format!("{} {}", r.status, r.body))); + } + n = match &hi { + None => n * 2, + Some((h, _)) if h - lo <= RESOLUTION => break, + Some((h, _)) => (lo + h) / 2, + }; + } + let Some((h, refusal)) = hi else { + return Err(fail( + format!("a refused DocumentAdd within {CAP} refs"), + format!("{lo} refs landed"), + )); + }; + ch.note(format!( + "{} bound: between {lo} and {h} refs, {} and {} bytes; at {h}: {refusal}", + ch.transport().label(), + lo * DOC_REF_BYTES, + h * DOC_REF_BYTES + )); + Ok(()) +} + +/// After the resume, the target finishes applying before the restore. +const APPLY_GRACE_MS: u64 = 10_000; + +/// The healthy request goes out this long after the pause has landed, +/// while the probe still waits on the relay's correlator. +const HEALTHY_AFTER_MS: u64 = 1_000; + +/// B4: the target is frozen mid-request (SIGSTOP, while it is still +/// reading or applying a big `DocumentAdd`), so the request is on its way +/// and no reply ever comes: the relay must give up with its correlator's +/// "response timeout", not a dial or stream error, and serve a healthy +/// target while that probe is still outstanding. The target is resumed +/// and its list restored either way. +pub(super) async fn b4(ch: &mut dyn Channel) -> Result<()> { + let (relay, silent, healthy) = (0, 1, 2); + let bounds = for_transport(ch.transport()); + let n = docs_for(bounds.silent_request); + let list = json!({ "Kind": "CollectionList" }); + let r = ch.send(relay, silent, Actor::Admin, list.clone()).await?; + expect_status(&r, 200, "CollectionList before the pause")?; + ch.control(Verb::PauseAfter { + node: silent, + delay_ms: bounds.pause_after_ms, + }) + .await?; + let (probe, next) = tokio::join!( + ch.send(relay, silent, Actor::Admin, document_op("DocumentAdd", n)), + async { + tokio::time::sleep(std::time::Duration::from_millis( + bounds.pause_after_ms + HEALTHY_AFTER_MS, + )) + .await; + ch.send(relay, healthy, Actor::Admin, list).await + } + ); + ch.control(Verb::Resume(silent)).await?; + tokio::time::sleep(std::time::Duration::from_millis(APPLY_GRACE_MS)).await; + let r = ch + .send( + relay, + silent, + Actor::Admin, + document_op("DocumentRemove", n), + ) + .await?; + expect_status(&r, 200, "DocumentRemove (restore)")?; + let r = ch + .send( + relay, + silent, + Actor::Admin, + json!({ "Kind": "DocumentList" }), + ) + .await?; + expect_status(&r, 200, "DocumentList after the restore")?; + let tracked = r.body["documents"].as_array().map_or(0, Vec::len); + if tracked != 0 { + return Err(fail( + "the target's document list empty after the restore", + format!("{tracked} tracked"), + )); + } + let probe = probe.map_err(|e| { + fail( + "a reply while the target is silent", + format!("no reply: {e:#}"), + ) + })?; + let body = probe.body.to_string(); + if probe.status != 400 || !body.contains("response timeout") { + return Err(fail( + "400 response timeout from the relay's correlator", + format!("{} after {} ms: {body}", probe.status, probe.latency_ms), + )); + } + expect_status( + &next?, + 200, + "CollectionList to a healthy target while the silent one hangs", + ) +} + +#[cfg(test)] +mod tests { + use std::cell::{Cell, RefCell}; + use std::rc::Rc; + + use super::super::cases::fake::*; + use super::super::cases::Outcome; + use super::super::client::Reply; + use super::*; + + fn replied(status: u16, ms: u64, body: &str) -> Result { + Ok(Reply { + status, + body: json!({ "error": body }), + latency_ms: ms, + }) + } + + fn refs(op: &Value) -> usize { + op["docs"].as_array().map_or(0, Vec::len) + } + + /// A target that lands `DocumentAdd` up to `max` refs and answers + /// `refusal` past it. + fn bounded(max: usize, refusal: Result) -> Fake { + let refusal = Cell::new(Some(refusal)); + Fake::new(move |_, _, _, _, op| match op["Kind"].as_str() { + Some("DocumentAdd" | "DocumentRemove") if refs(op) > max => refusal + .take() + .unwrap_or_else(|| replied(400, 30_000, "response timeout")), + _ => admin_view(op), + }) + } + + #[test] + fn bounds_follow_the_transport() { + assert_eq!( + for_transport(Transport::Libp2p).max_request, + 16 * 1024 * 1024 + ); + assert_eq!(for_transport(Transport::Iroh).max_request, 4 * 1024 * 1024); + } + + #[test] + fn document_op_refs_are_well_formed_and_sized() { + let op = document_op("DocumentAdd", 3); + assert_eq!( + op["docs"][2]["doc_id"], + "bae-00000000-0000-0000-0000-000000000002" + ); + assert!(docs_for(LIBP2P.max_request) * DOC_REF_BYTES <= LIBP2P.max_request); + assert!(docs_for(MARGIN) > 0); + } + + #[tokio::test] + async fn b1_lands_under_the_bound_and_removes_again() { + let (tx, rx) = std::sync::mpsc::channel(); + let outcome = run_one("B1", move |_, _, _, _, op| { + tx.send((op["Kind"].as_str().unwrap().to_string(), refs(op))) + .unwrap(); + admin_view(op) + }) + .await; + assert_eq!(outcome, Outcome::Pass); + let n = docs_for(LIBP2P.max_request - MARGIN); + let seen: Vec<_> = rx + .try_iter() + .skip_while(|s| s.0 == "CollectionList") + .collect(); + assert_eq!(seen[0], ("DocumentAdd".to_string(), n)); + assert!(seen.contains(&("DocumentRemove".to_string(), n))); + + let (tx, rx) = std::sync::mpsc::channel(); + let mut fake = Fake::new(move |_, _, _, _, op| { + tx.send(refs(op)).unwrap(); + admin_view(op) + }); + fake.transport = Transport::Iroh; + assert_eq!(run_fake("B1", fake).await.0, Outcome::Pass); + assert_eq!( + rx.try_iter().find(|n| *n > 0), + Some(docs_for(IROH.max_request - MARGIN)) + ); + + let (outcome, _) = run_fake( + "B1", + bounded(n - 1, replied(400, 30_000, "response timeout")), + ) + .await; + assert!( + matches!(&outcome, Outcome::Fail { expected, got } if expected.contains("under the bound") && got.starts_with("400")), + "{outcome:?}" + ); + } + + #[tokio::test] + async fn b2_needs_a_refusal_then_the_target_then_the_relay() { + let n = docs_for(LIBP2P.max_request + MARGIN); + let (outcome, notes) = run_noted( + "B2", + bounded(n - 1, replied(400, 30_000, "response timeout")), + ) + .await; + assert_eq!(outcome, Outcome::Pass); + assert_eq!(notes.len(), 1); + assert!(notes[0].contains("400 after 30000 ms"), "{notes:?}"); + + let served = run_fake("B2", bounded(n + 1, status(200))).await.0; + assert!( + matches!(&served, Outcome::Fail { expected, .. } if expected.contains("clean refusal")), + "{served:?}" + ); + let crashed = run_fake("B2", bounded(n - 1, status(502))).await.0; + assert!( + matches!(&crashed, Outcome::Fail { got, .. } if got.starts_with("502")), + "{crashed:?}" + ); + + for (down, what) in [(1, "on the target"), (2, "to a third node")] { + let mut added = false; + let outcome = run_one("B2", move |_, target, _, _, op| { + if op["Kind"] == "DocumentAdd" { + added = true; + replied(400, 30_000, "response timeout") + } else if target == down && added { + replied(400, 10_000, "dial timeout") + } else { + admin_view(op) + } + }) + .await; + assert!( + matches!(&outcome, Outcome::Fail { expected, .. } if expected.contains(what)), + "{what}: {outcome:?}" + ); + } + } + + #[tokio::test] + async fn b3_brackets_the_bound_and_restores_what_landed() { + let (tx, rx) = std::sync::mpsc::channel(); + let mut fake = bounded(100_000, replied(400, 30_000, "response timeout")); + let inner = RefCell::new(fake.rule.replace(Box::new(|_, _, _, _, _| status(200)))); + fake.rule = RefCell::new(Box::new(move |r, t, a, actor, op| { + tx.send((op["Kind"].as_str().unwrap().to_string(), refs(op))) + .unwrap(); + (inner.borrow_mut())(r, t, a, actor, op) + })); + let (outcome, notes) = run_noted("B3", fake).await; + assert_eq!(outcome, Outcome::Pass); + let last = notes.last().unwrap(); + assert!(last.starts_with("libp2p bound: between "), "{last}"); + let words: Vec<&str> = last.split(' ').collect(); + let (lo, hi): (usize, usize) = (words[3].parse().unwrap(), words[5].parse().unwrap()); + assert!( + lo <= 100_000 && 100_000 < hi && hi - lo <= RESOLUTION, + "{last}" + ); + assert!( + last.ends_with("400 {\"error\":\"response timeout\"}"), + "{last}" + ); + let ops: Vec<_> = rx.try_iter().collect(); + for (kind, n) in &ops { + if kind == "DocumentAdd" && *n <= 100_000 { + assert!( + ops.contains(&("DocumentRemove".into(), *n)), + "{n} not removed" + ); + } + } + assert!( + !ops.contains(&("DocumentRemove".into(), 131072)), + "refused probe removed" + ); + } + + #[tokio::test] + async fn b3_fails_when_nothing_is_refused_under_the_cap() { + let (outcome, _) = run_fake("B3", bounded(usize::MAX, status(200))).await; + assert!( + matches!(&outcome, Outcome::Fail { expected, .. } if expected.contains("refused")), + "{outcome:?}" + ); + } + + const PAUSE: Verb = Verb::PauseAfter { + node: 1, + delay_ms: LIBP2P.pause_after_ms, + }; + + /// Frozen: node 1 answers `reply` to the one request in flight while + /// its pause is pending; everyone else is healthy. + fn paused_target(reply: Result) -> Fake { + let verbs: Rc>> = Default::default(); + let seen = verbs.clone(); + let reply = Cell::new(Some(reply)); + let mut fake = Fake::new(move |_, target, _, _, op| { + if target == 1 && seen.borrow().last() == Some(&PAUSE) { + return reply.take().expect("one probe while paused"); + } + admin_view(op) + }); + fake.verbs = verbs; + fake + } + + #[tokio::test(start_paused = true)] + async fn b4_wants_the_correlator_timeout_not_a_dial_error_and_resumes() { + let (outcome, verbs) = run_fake( + "B4", + paused_target(replied(400, 30_050, "response timeout")), + ) + .await; + assert_eq!(outcome, Outcome::Pass); + assert_eq!(verbs, [PAUSE, Verb::Resume(1)]); + + let dial = run_fake("B4", paused_target(replied(400, 10_000, "dial timeout"))).await; + assert!( + matches!(&dial.0, Outcome::Fail { expected, got } if expected.contains("response timeout") && got.contains("dial timeout")), + "{:?}", + dial.0 + ); + assert_eq!(dial.1, [PAUSE, Verb::Resume(1)]); + + let hung = run_fake("B4", paused_target(Err(eyre::eyre!("operation timed out")))).await; + assert!( + matches!(&hung.0, Outcome::Fail { got, .. } if got.contains("timed out")), + "{:?}", + hung.0 + ); + assert_eq!(hung.1, [PAUSE, Verb::Resume(1)]); + } + + #[tokio::test(start_paused = true)] + async fn b4_asks_the_healthy_target_while_the_probe_is_outstanding() { + let t0 = tokio::time::Instant::now(); + let (tx, rx) = std::sync::mpsc::channel(); + let mut fake = paused_target(replied(400, 30_050, "response timeout")); + fake.timed = true; + let inner = RefCell::new(fake.rule.replace(Box::new(|_, _, _, _, _| status(200)))); + fake.rule = RefCell::new(Box::new(move |r, t, a, actor, op| { + if t == 2 { + tx.send(t0.elapsed().as_millis() as u64).unwrap(); + } + (inner.borrow_mut())(r, t, a, actor, op) + })); + let (outcome, _) = run_fake("B4", fake).await; + assert_eq!(outcome, Outcome::Pass); + let asked = rx.try_iter().last().unwrap(); + assert!( + asked > LIBP2P.pause_after_ms && asked < 30_050, + "asked at {asked} ms" + ); + } + + #[tokio::test(start_paused = true)] + async fn b4_fails_when_the_healthy_target_is_not_served_or_the_restore_leaves_docs() { + let verbs: Rc>> = Default::default(); + let seen = verbs.clone(); + let mut fake = Fake::new( + move |_, target, _, _, op| match (target, seen.borrow().last()) { + (1 | 2, Some(&PAUSE)) => replied(400, 30_050, "response timeout"), + _ => admin_view(op), + }, + ); + fake.verbs = verbs; + let (outcome, verbs) = run_fake("B4", fake).await; + assert!( + matches!(&outcome, Outcome::Fail { expected, .. } if expected.contains("healthy target")), + "{outcome:?}" + ); + assert_eq!(verbs, [PAUSE, Verb::Resume(1)]); + + let mut fake = paused_target(replied(400, 30_050, "response timeout")); + let inner = RefCell::new(fake.rule.replace(Box::new(|_, _, _, _, _| status(200)))); + fake.rule = RefCell::new(Box::new(move |r, t, a, actor, op| { + if op["Kind"] == "DocumentList" { + return ok(json!({"Kind": "Documents", "documents": [{"doc_id": "bae-1"}]})); + } + (inner.borrow_mut())(r, t, a, actor, op) + })); + let (outcome, _) = run_fake("B4", fake).await; + assert!( + matches!(&outcome, Outcome::Fail { expected, got } if expected.contains("empty after the restore") && got == "1 tracked"), + "{outcome:?}" + ); + } +} diff --git a/crates/soak/src/manage/cases.rs b/crates/soak/src/manage/cases.rs new file mode 100644 index 0000000..f2e060f --- /dev/null +++ b/crates/soak/src/manage/cases.rs @@ -0,0 +1,826 @@ +//! The case table and its runner. A case is one function with one +//! expectation; it speaks to the cluster only through [`Channel`], so the +//! runner and every case run against a scripted fake in the unit tests. +//! Each case restores what it changed. The cases live by group in +//! `routing.rs`, `authz.rs`, `state.rs`, `bounds.rs`, `partition.rs` and +//! `hybrid.rs`. + +use std::collections::HashMap; +use std::fmt; + +use eyre::{eyre, Result}; +use futures::future::{BoxFuture, LocalBoxFuture}; +use serde::Serialize; +use serde_json::{json, Value}; + +use super::actors::Actor; +use super::client::Reply; +use super::data::list_users; +use super::{authz, bounds, hybrid, partition, routing, state}; +use crate::Transport; + +pub const COLLECTION: &str = "User"; + +/// One relayed request as the report records it. +#[derive(Clone, Debug, Serialize)] +pub struct OpRecord { + pub relay: usize, + pub target: usize, + pub actor: Actor, + pub kind: String, + pub status: u16, + pub latency_ms: u64, + /// The target's list for the op's family after a mutate, as admin. + pub target_state: Option, +} + +/// A cluster verb a case needs beyond sending; the live channel maps each +/// to the harness, the fake records it. +#[derive(Clone, Debug, PartialEq, Eq)] +pub enum Verb { + Stop(usize), + Start(usize), + /// The actors' grants on one node again, after a restart. + Regrant(usize), + Grant { + node: usize, + actor: Actor, + relation: &'static str, + }, + Revoke { + node: usize, + actor: Actor, + relation: &'static str, + }, + /// NAC on or off on one node, as the owner. + Nac { + node: usize, + on: bool, + }, + Partition(usize), + Rejoin(usize), + /// Freeze the node's process `delay_ms` from now, without waiting: its + /// connections stay up, it reads nothing until `Resume`. + PauseAfter { + node: usize, + delay_ms: u64, + }, + Resume(usize), + /// A replicator from `node` to `peer` through `node`'s own HTTP API as + /// the owner, with the suite's collection and these filters. + LocalReplicatorAdd { + node: usize, + peer: usize, + filters: Value, + }, +} + +/// Relay `op` through node `relay` to node `target` as `actor`. Sends +/// borrow the channel shared, so a case can hold two in flight at once. +pub trait Channel { + /// The Rust nodes, `0..len`: the only ones a manage op can target. + fn len(&self) -> usize; + /// The Go nodes, replication peers only; their own HTTP still answers + /// `gql` and `replicators`. + fn go_nodes(&self) -> Vec { + Vec::new() + } + fn transport(&self) -> Transport; + fn addr(&self, node: usize) -> String; + fn peer_id(&self, node: usize) -> String; + /// `send` with the token minted for `audience` instead of the target. + fn send_for<'a>( + &'a self, + relay: usize, + target: usize, + audience: usize, + actor: Actor, + op: Value, + ) -> LocalBoxFuture<'a, Result>; + fn send<'a>( + &'a self, + relay: usize, + target: usize, + actor: Actor, + op: Value, + ) -> LocalBoxFuture<'a, Result> { + self.send_for(relay, target, target, actor, op) + } + fn control<'a>(&'a mut self, verb: Verb) -> LocalBoxFuture<'a, Result<()>>; + /// GraphQL on `node`'s own HTTP API as the owner; the `data` object. + /// Owned so a case can spawn it beside its ops. + fn gql(&self, node: usize, query: String) -> BoxFuture<'static, Result>; + /// `node`'s replicator list over its own HTTP API as the owner, in the + /// runtime's own shape; `Err` while the node is not up. + fn replicators(&self, node: usize) -> Result; + fn can_partition(&self) -> bool; + /// Something the report keeps beside the outcome: a mode the case + /// observed and did not assert. + fn note(&mut self, text: String); + /// The requests since the last call, for the report. + fn take_records(&mut self) -> Vec { + Vec::new() + } + fn take_notes(&mut self) -> Vec { + Vec::new() + } + /// Rows a case ran on the runner's behalf (H1); they precede its own. + fn embed(&mut self, _reports: Vec) {} + fn take_embedded(&mut self) -> Vec { + Vec::new() + } +} + +#[derive(Clone, Debug, PartialEq, Eq, Serialize)] +#[serde(tag = "outcome")] +pub enum Outcome { + Pass, + Fail { expected: String, got: String }, + Skip { reason: String }, + Infra { error: String }, +} + +/// An expectation miss, distinct from a harness fault: the runner turns it +/// into [`Outcome::Fail`] and any other error into [`Outcome::Infra`]. +#[derive(Debug)] +pub struct Failed { + pub expected: String, + pub got: String, +} + +impl fmt::Display for Failed { + fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { + write!(f, "expected {}, got {}", self.expected, self.got) + } +} + +impl std::error::Error for Failed {} + +pub(super) fn fail(expected: impl Into, got: impl Into) -> eyre::Report { + Failed { + expected: expected.into(), + got: got.into(), + } + .into() +} + +/// A case that cannot run here; the runner turns it into [`Outcome::Skip`]. +#[derive(Debug)] +pub struct Skipped(pub String); + +impl fmt::Display for Skipped { + fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { + write!(f, "skipped: {}", self.0) + } +} + +impl std::error::Error for Skipped {} + +pub(super) fn skip(reason: impl Into) -> eyre::Report { + Skipped(reason.into()).into() +} + +pub(super) fn expect_status(r: &Reply, want: u16, what: &str) -> Result<()> { + if r.status == want { + Ok(()) + } else { + Err(fail( + format!("{want} on {what}"), + format!("{} {}", r.status, r.body), + )) + } +} + +/// Minimum Rust node count a case needs; the case uses nodes `0..min_rust`. +#[derive(Clone, Copy, Debug)] +pub struct Topo { + pub min_rust: usize, +} + +pub struct Case { + pub name: &'static str, + pub requires: Topo, + pub run: for<'a> fn(&'a mut dyn Channel) -> LocalBoxFuture<'a, Result<()>>, +} + +#[derive(Serialize)] +pub struct CaseReport { + pub name: &'static str, + pub outcome: Outcome, + pub ops: Vec, + pub notes: Vec, +} + +/// Cases a default selection leaves out; `--locate-size-bound` adds B3. +pub const OPT_IN: &[&str] = &["B3"]; + +/// The table, in the default order: the bounds group last, so a target +/// it wedges cannot poison the state, partition or concurrency cases. +pub fn all() -> Vec { + let two = Topo { min_rust: 2 }; + let three = Topo { min_rust: 3 }; + vec![ + Case { + name: "R1", + requires: three, + run: |ch| Box::pin(routing::r1(ch)), + }, + Case { + name: "R2", + requires: two, + run: |ch| Box::pin(routing::r2(ch)), + }, + Case { + name: "R3", + requires: two, + run: |ch| Box::pin(routing::r3(ch)), + }, + Case { + name: "A1", + requires: two, + run: |ch| Box::pin(authz::a1(ch)), + }, + Case { + name: "A2", + requires: two, + run: |ch| Box::pin(authz::a2(ch)), + }, + Case { + name: "A3", + requires: two, + run: |ch| Box::pin(authz::a3(ch)), + }, + Case { + name: "A4", + requires: two, + run: |ch| Box::pin(authz::a4(ch)), + }, + Case { + name: "A5", + requires: three, + run: |ch| Box::pin(authz::a5(ch)), + }, + Case { + name: "A6", + requires: two, + run: |ch| Box::pin(authz::a6(ch)), + }, + Case { + name: "S1", + requires: two, + run: |ch| Box::pin(state::s1(ch)), + }, + Case { + name: "S2", + requires: two, + run: |ch| Box::pin(state::s2(ch)), + }, + Case { + name: "S3", + requires: two, + run: |ch| Box::pin(state::s3(ch)), + }, + Case { + name: "S4", + requires: two, + run: |ch| Box::pin(state::s4(ch)), + }, + Case { + name: "P1", + requires: two, + run: |ch| Box::pin(partition::p1(ch)), + }, + Case { + name: "C1", + requires: two, + run: |ch| Box::pin(partition::c1(ch)), + }, + Case { + name: "H1", + requires: two, + run: |ch| Box::pin(hybrid::h1(ch)), + }, + Case { + name: "B1", + requires: two, + run: |ch| Box::pin(bounds::b1(ch)), + }, + Case { + name: "B2", + requires: three, + run: |ch| Box::pin(bounds::b2(ch)), + }, + Case { + name: "B3", + requires: two, + run: |ch| Box::pin(bounds::b3(ch)), + }, + Case { + name: "B4", + requires: three, + run: |ch| Box::pin(bounds::b4(ch)), + }, + ] +} + +/// `--cases S1,R2` in the order given; `None` is every case but [`OPT_IN`] +/// in table order. +pub fn select<'a>(all: &'a [Case], filter: Option<&str>) -> Result> { + let Some(filter) = filter else { + return Ok(all.iter().filter(|c| !OPT_IN.contains(&c.name)).collect()); + }; + filter + .split(',') + .map(str::trim) + .map(|w| { + all.iter() + .find(|c| c.name == w) + .ok_or_else(|| eyre!("--cases: unknown case {w}")) + }) + .collect() +} + +/// Before a case, the cheapest admin query to every node it uses: node 0 +/// and every Go node over their own HTTP as the owner, the other Rust +/// nodes through node 0 as relay. A node that does not answer is named +/// with the last case that used it, so a wedge left behind reads as +/// `Infra`, not as this case's `Fail`. +async fn unreachable( + ch: &mut dyn Channel, + case: &Case, + last_touch: &HashMap, +) -> Option { + let go = ch.go_nodes(); + for node in (0..case.requires.min_rust).chain(go.iter().copied()) { + let answered = if node == 0 || go.contains(&node) { + ch.gql(node, list_users()) + .await + .map(drop) + .map_err(|e| format!("{e:#}")) + } else { + match ch + .send(0, node, Actor::Admin, json!({ "Kind": "CollectionList" })) + .await + { + Ok(r) if r.status == 200 => Ok(()), + Ok(r) => Err(format!( + "{} after {} ms: {}", + r.status, r.latency_ms, r.body + )), + Err(e) => Err(format!("{e:#}")), + } + }; + if let Err(why) = answered { + ch.take_records(); + let runtime = if go.contains(&node) { "go " } else { "" }; + return Some(format!( + "{runtime}node {node} unreachable before {}; last case to touch it: {} ({why})", + case.name, + last_touch.get(&node).unwrap_or(&"none") + )); + } + } + ch.take_records(); + None +} + +pub async fn run_all(ch: &mut dyn Channel, cases: &[&Case], rust_nodes: usize) -> Vec { + let mut reports = Vec::new(); + let mut last_touch: HashMap = HashMap::new(); + for case in cases { + let outcome = if rust_nodes < case.requires.min_rust { + Outcome::Skip { + reason: format!( + "needs {} Rust nodes, topology has {rust_nodes}", + case.requires.min_rust + ), + } + } else if let Some(error) = unreachable(ch, case, &last_touch).await { + Outcome::Infra { error } + } else { + for node in (0..case.requires.min_rust).chain(ch.go_nodes()) { + last_touch.insert(node, case.name); + } + match (case.run)(ch).await { + Ok(()) => Outcome::Pass, + Err(e) => match (e.downcast_ref::(), e.downcast_ref::()) { + (Some(f), _) => Outcome::Fail { + expected: f.expected.clone(), + got: f.got.clone(), + }, + (_, Some(s)) => Outcome::Skip { + reason: s.0.clone(), + }, + _ => Outcome::Infra { + error: format!("{e:#}"), + }, + }, + } + }; + let notes = ch.take_notes(); + println!("case {}: {outcome:?} {}", case.name, notes.join("; ")); + reports.extend(ch.take_embedded()); + reports.push(CaseReport { + name: case.name, + outcome, + ops: ch.take_records(), + notes, + }); + } + reports +} + +pub(super) fn replicator_add(addr: &str) -> Value { + json!({ "Kind": "ReplicatorAdd", "addresses": [addr], "collection_ids": [COLLECTION] }) +} + +pub(super) fn replicator_delete(addr: &str) -> Value { + json!({ "Kind": "ReplicatorDelete", "addresses": [addr], "collection_ids": [COLLECTION] }) +} + +pub(super) fn collection_add() -> Value { + json!({ "Kind": "CollectionAdd", "collection_ids": [COLLECTION] }) +} + +pub(super) fn collection_remove() -> Value { + json!({ "Kind": "CollectionRemove", "collection_ids": [COLLECTION] }) +} + +/// Entries of a `Replicators` reply whose peer is `peer_id`. The relayed +/// body is the http crate's snake_case `ReplicatorInfo` (`id`, `address`, +/// `collections` as collection ids), not the p2p wire type. +pub(super) fn replicators_for(body: &Value, peer_id: &str) -> usize { + body["replicators"] + .as_array() + .map_or(0, |a| a.iter().filter(|r| r["id"] == peer_id).count()) +} + +/// Every mutate and query op, with a payload that deserializes at the relay. +pub(super) fn every_op(relay_addr: &str) -> Vec { + let doc = + json!({ "collection": COLLECTION, "doc_id": "bae-00000000-0000-0000-0000-000000000000" }); + vec![ + replicator_add(relay_addr), + replicator_delete(relay_addr), + json!({ "Kind": "CollectionAdd", "collection_ids": [COLLECTION] }), + json!({ "Kind": "CollectionRemove", "collection_ids": [COLLECTION] }), + json!({ "Kind": "DocumentAdd", "docs": [doc] }), + json!({ "Kind": "DocumentRemove", "docs": [doc] }), + json!({ "Kind": "PeerConnect", "address": relay_addr }), + json!({ "Kind": "PeerDisconnect", "address": relay_addr }), + json!({ "Kind": "ReplicatorList" }), + json!({ "Kind": "CollectionList" }), + json!({ "Kind": "DocumentList" }), + ] +} + +/// The target's managed state as admin: replicators as (peer, collections, +/// filters), subscriptions, tracked documents. Connection health fields +/// are left out so an idle status flip does not read as a change. +pub(super) async fn managed_state( + ch: &mut dyn Channel, + relay: usize, + target: usize, +) -> Result { + let mut out = Vec::new(); + for kind in ["ReplicatorList", "CollectionList", "DocumentList"] { + let r = ch + .send(relay, target, Actor::Admin, json!({ "Kind": kind })) + .await?; + expect_status(&r, 200, &format!("{kind} as admin"))?; + out.push(r.body); + } + let mut replicators: Vec = out[0]["replicators"] + .as_array() + .into_iter() + .flatten() + .map(|r| json!([r["id"], r["collections"], r["filters"]])) + .collect(); + replicators.sort_by_key(|v| v.to_string()); + Ok(json!({ + "replicators": replicators, + "collections": out[1]["values"], + "documents": out[2]["documents"], + })) +} + +/// A scripted channel for the case tests in every group. +#[cfg(test)] +pub(super) mod fake { + use super::*; + + use std::cell::RefCell; + use std::rc::Rc; + + /// `(relay, target, audience, actor, op)`. + pub type Rule = dyn FnMut(usize, usize, usize, Actor, &Value) -> Result; + + /// `(node, query)`. + pub type GqlRule = dyn FnMut(usize, &str) -> Result; + + /// `(node)`: the node's own replicator list. + pub type OwnRule = dyn FnMut(usize) -> Result; + + /// A channel that answers from a rule and records every verb; `Err` + /// from the rule is a transport fault. Share `verbs` with the rule + /// when a reply depends on a verb (a stopped node, a revoked grant). + /// With `timed`, a reply arrives its `latency_ms` later in tokio time. + /// `gql` answers the data plane `gql_ms` later, in tokio time; the + /// default is an empty node answering at once. `own` answers a node's + /// own replicator list; the default is an empty list. + pub struct Fake { + pub rule: RefCell>, + pub timed: bool, + pub gql: RefCell>, + pub gql_ms: u64, + pub own: RefCell>, + pub verbs: Rc>>, + pub notes: Vec, + pub partition: bool, + pub transport: Transport, + pub go: Vec, + pub embedded: Vec, + } + + impl Fake { + pub fn new( + rule: impl FnMut(usize, usize, usize, Actor, &Value) -> Result + 'static, + ) -> Self { + Self { + rule: RefCell::new(Box::new(rule)), + timed: false, + gql: RefCell::new(Box::new(|_, _| Ok(json!({ "User": [] })))), + gql_ms: 0, + own: RefCell::new(Box::new(|_| Ok(json!([])))), + verbs: Rc::default(), + notes: Vec::new(), + partition: true, + transport: Transport::Libp2p, + go: Vec::new(), + embedded: Vec::new(), + } + } + } + + impl Channel for Fake { + fn len(&self) -> usize { + 3 + } + fn go_nodes(&self) -> Vec { + self.go.clone() + } + fn transport(&self) -> Transport { + self.transport + } + fn addr(&self, node: usize) -> String { + format!("/ip4/127.0.0.1/tcp/{node}/p2p/peer{node}") + } + fn peer_id(&self, node: usize) -> String { + format!("peer{node}") + } + fn send_for<'a>( + &'a self, + relay: usize, + target: usize, + audience: usize, + actor: Actor, + op: Value, + ) -> LocalBoxFuture<'a, Result> { + let r = (self.rule.borrow_mut())(relay, target, audience, actor, &op); + let delay = r + .as_ref() + .ok() + .filter(|_| self.timed) + .map(|r| std::time::Duration::from_millis(r.latency_ms)); + Box::pin(async move { + if let Some(delay) = delay { + tokio::time::sleep(delay).await; + } + r + }) + } + fn control<'a>(&'a mut self, verb: Verb) -> LocalBoxFuture<'a, Result<()>> { + self.verbs.borrow_mut().push(verb); + Box::pin(async { Ok(()) }) + } + fn gql(&self, node: usize, query: String) -> BoxFuture<'static, Result> { + let r = (self.gql.borrow_mut())(node, &query); + let delay = std::time::Duration::from_millis(self.gql_ms); + Box::pin(async move { + if !delay.is_zero() { + tokio::time::sleep(delay).await; + } + r + }) + } + fn replicators(&self, node: usize) -> Result { + (self.own.borrow_mut())(node) + } + fn can_partition(&self) -> bool { + self.partition + } + fn note(&mut self, text: String) { + self.notes.push(text); + } + fn take_notes(&mut self) -> Vec { + std::mem::take(&mut self.notes) + } + fn embed(&mut self, reports: Vec) { + self.embedded.extend(reports); + } + fn take_embedded(&mut self) -> Vec { + std::mem::take(&mut self.embedded) + } + } + + pub fn ok(body: Value) -> Result { + Ok(Reply { + status: 200, + body, + latency_ms: 1, + }) + } + + pub fn status(code: u16) -> Result { + Ok(Reply { + status: code, + body: Value::Null, + latency_ms: 1, + }) + } + + /// Admin sees a healthy mesh: node 1 replicates to node 0, subscribes to + /// the collection, tracks no documents. + pub fn admin_view(op: &Value) -> Result { + match op["Kind"].as_str().unwrap() { + "ReplicatorList" => ok(json!({"Kind": "Replicators", "replicators": [ + {"id": "peer0", "address": "/ip4/127.0.0.1/tcp/0/p2p/peer0", "collections": ["bafy-user"]} + ]})), + "CollectionList" => ok(json!({"Kind": "Strings", "values": [COLLECTION]})), + "DocumentList" => ok(json!({"Kind": "Documents", "documents": []})), + _ => status(200), + } + } + + pub fn by_name(name: &str) -> Case { + all().into_iter().find(|c| c.name == name).unwrap() + } + + /// Run `name` on a three-node fake; the outcome and the verbs it used. + pub async fn run_fake(name: &str, mut fake: Fake) -> (Outcome, Vec) { + let case = by_name(name); + let outcome = run_all(&mut fake, &[&case], 3).await.remove(0).outcome; + let verbs = fake.verbs.borrow().clone(); + (outcome, verbs) + } + + /// `run_fake`, with the notes the case recorded. + pub async fn run_noted(name: &str, mut fake: Fake) -> (Outcome, Vec) { + let case = by_name(name); + let report = run_all(&mut fake, &[&case], 3).await.remove(0); + (report.outcome, report.notes) + } + + pub async fn run_one( + name: &str, + rule: impl FnMut(usize, usize, usize, Actor, &Value) -> Result + 'static, + ) -> Outcome { + run_fake(name, Fake::new(rule)).await.0 + } +} + +#[cfg(test)] +mod tests { + use super::fake::*; + use super::*; + + #[tokio::test] + async fn runner_classifies_pass_fail_infra_and_skip() { + assert_eq!( + run_one("R2", |_, _, _, _, op| admin_view(op)).await, + Outcome::Pass + ); + let failed = run_one("R2", |_, _, _, _, op| { + if op["Kind"] == "CollectionAdd" { + status(400) + } else { + admin_view(op) + } + }) + .await; + assert!( + matches!(&failed, Outcome::Fail { expected, got } if expected.starts_with("200") && got.starts_with("400")), + "{failed:?}" + ); + let infra = run_one("R2", |_, _, _, _, _| eyre::bail!("connection refused")).await; + assert!( + matches!(&infra, Outcome::Infra { error } if error.contains("refused")), + "{infra:?}" + ); + + let mut fake = Fake::new(|_, _, _, _, op| admin_view(op)); + let three = Case { + name: "X", + requires: Topo { min_rust: 3 }, + run: |ch| Box::pin(routing::r2(ch)), + }; + let r = run_all(&mut fake, &[&three], 2).await.remove(0); + assert!(matches!(r.outcome, Outcome::Skip { .. }), "{:?}", r.outcome); + + let bowed_out = Case { + name: "Y", + requires: Topo { min_rust: 2 }, + run: |_| Box::pin(async { Err(skip("no partition here")) }), + }; + let r = run_all(&mut fake, &[&bowed_out], 3).await.remove(0); + assert_eq!( + r.outcome, + Outcome::Skip { + reason: "no partition here".into() + } + ); + } + + #[tokio::test] + async fn runner_reports_an_unreachable_node_as_infra_naming_the_last_case_to_touch_it() { + // Node 1 stops answering once R2's restoring CollectionRemove has landed. + let wedged = std::cell::Cell::new(false); + let mut fake = Fake::new(move |_, target, _, _, op| { + if target == 1 && wedged.get() { + return Ok(Reply { + status: 400, + body: json!({"error": "dial error: timed out"}), + latency_ms: 30_012, + }); + } + wedged.set(wedged.get() || op["Kind"] == "CollectionRemove"); + admin_view(op) + }); + let cases = [by_name("R2"), by_name("S1")]; + let reports = run_all(&mut fake, &[&cases[0], &cases[1]], 3).await; + assert_eq!(reports[0].outcome, Outcome::Pass); + assert_eq!( + reports[1].outcome, + Outcome::Infra { + error: "node 1 unreachable before S1; last case to touch it: R2 (400 after 30012 ms: {\"error\":\"dial error: timed out\"})".into() + } + ); + + let untouched = run_one("S1", |_, _, _, _, _| eyre::bail!("connection refused")).await; + assert_eq!( + untouched, + Outcome::Infra { + error: + "node 1 unreachable before S1; last case to touch it: none (connection refused)" + .into() + } + ); + } + + #[tokio::test] + async fn runner_probes_the_go_nodes_over_their_own_http_and_names_one_that_is_down() { + let mut fake = Fake::new(|_, _, _, _, op| admin_view(op)); + fake.go = vec![3, 4]; + fake.gql = std::cell::RefCell::new(Box::new(|node, _| { + if node == 4 { + eyre::bail!("connection refused") + } + Ok(json!({ "User": [] })) + })); + let cases = [by_name("R2"), by_name("S1")]; + let reports = run_all(&mut fake, &[&cases[0], &cases[1]], 3).await; + assert_eq!( + reports[0].outcome, + Outcome::Infra { + error: "go node 4 unreachable before R2; last case to touch it: none (connection refused)".into() + } + ); + assert_eq!( + reports[1].outcome, + Outcome::Infra { + error: "go node 4 unreachable before S1; last case to touch it: none (connection refused)".into() + } + ); + + let mut fake = Fake::new(|_, _, _, _, op| admin_view(op)); + fake.go = vec![3]; + let (outcome, _) = run_fake("R2", fake).await; + assert_eq!(outcome, Outcome::Pass); + } + + #[test] + fn select_runs_bounds_last_by_default_keeps_the_given_order_and_rejects_unknown_names() { + let table = all(); + let names = |v: Vec<&Case>| v.iter().map(|c| c.name).collect::>(); + assert_eq!( + names(select(&table, None).unwrap()), + [ + "R1", "R2", "R3", "A1", "A2", "A3", "A4", "A5", "A6", "S1", "S2", "S3", "S4", "P1", + "C1", "H1", "B1", "B2", "B4" + ] + ); + assert_eq!(names(select(&table, Some("S1, R2")).unwrap()), ["S1", "R2"]); + assert_eq!(names(select(&table, Some("B3")).unwrap()), ["B3"]); + assert!(select(&table, Some("R2,Z9")).is_err()); + } +} diff --git a/crates/soak/src/manage/client.rs b/crates/soak/src/manage/client.rs new file mode 100644 index 0000000..fc1b9be --- /dev/null +++ b/crates/soak/src/manage/client.rs @@ -0,0 +1,57 @@ +//! One raw request on the P2P management channel: POST to a relay node's +//! `/api/v0/p2p/manage` (or `/manage/query` for the `*List` kinds) with +//! `{Target, AuthToken, Op}`. No retries, no interpretation (copy of the +//! integration tests' `post_manage`, tools/integration-test/tests/manage_relay_common.rs). + +use std::time::Instant; + +use eyre::{Result, WrapErr}; +use serde::Serialize; +use serde_json::{json, Value}; + +use crate::auth::auth_token; + +#[derive(Clone, Debug, Serialize)] +pub struct Reply { + pub status: u16, + pub body: Value, + pub latency_ms: u64, +} + +/// The `*List` kinds go to `/manage/query`; everything else mutates. +pub fn is_query(op: &Value) -> bool { + op["Kind"].as_str().is_some_and(|k| k.ends_with("List")) +} + +/// `courier_key` is the HTTP caller at the relay (needs `connect-p2p-peer` +/// there); `actor_token` is the relayed identity the target authorizes. +pub async fn post( + http: &reqwest::Client, + relay_url: &str, + courier_key: &str, + target_addr: &str, + actor_token: &str, + op: &Value, +) -> Result { + let route = if is_query(op) { + "/api/v0/p2p/manage/query" + } else { + "/api/v0/p2p/manage" + }; + let body = json!({ "Target": target_addr, "AuthToken": actor_token, "Op": op }); + let started = Instant::now(); + let resp = http + .post(format!("{relay_url}{route}")) + .bearer_auth(auth_token(courier_key, relay_url)?) + .json(&body) + .send() + .await + .wrap_err_with(|| format!("POST {route} at {relay_url}"))?; + let status = resp.status().as_u16(); + let text = resp.text().await.unwrap_or_default(); + Ok(Reply { + status, + body: serde_json::from_str(&text).unwrap_or(Value::String(text)), + latency_ms: started.elapsed().as_millis() as u64, + }) +} diff --git a/crates/soak/src/manage/data.rs b/crates/soak/src/manage/data.rs new file mode 100644 index 0000000..310dd94 --- /dev/null +++ b/crates/soak/src/manage/data.rs @@ -0,0 +1,209 @@ +//! The data plane the state, bounds and concurrency cases lean on: +//! documents in the suite's collection, written and listed over a node's +//! own HTTP API as the owner, and a wait for them to reach another node. + +use std::sync::atomic::{AtomicUsize, Ordering}; +use std::time::Duration; + +use eyre::Result; +use serde_json::Value; + +use super::cases::{Channel, COLLECTION}; + +/// Polls of `converge`, half a second apart. +const SETTLE_POLLS: usize = 40; + +/// Ids are content-derived, so every document gets a name of its own. +static NEXT_NAME: AtomicUsize = AtomicUsize::new(0); + +pub(super) fn create_users(n: usize, age: i64) -> String { + let first = NEXT_NAME.fetch_add(n, Ordering::Relaxed); + let inputs: Vec = (first..first + n) + .map(|i| format!("{{name: \"u{i}\", age: {age}}}")) + .collect(); + format!( + "mutation {{ add_{COLLECTION}(input: [{}]) {{ _docID }} }}", + inputs.join(", ") + ) +} + +/// `age` is immutable (a filter field); an update touches `name`. +pub(super) fn update_user(id: &str, name: &str) -> String { + format!( + "mutation {{ update_{COLLECTION}(docID: \"{id}\", input: {{name: \"{name}\"}}) {{ _docID }} }}" + ) +} + +pub(super) fn list_users() -> String { + format!("{{ {COLLECTION} {{ _docID name }} }}") +} + +/// The `_docID`s under `key` of a `data` object. +pub(super) fn ids_in(data: &Value, key: &str) -> Vec { + data[key] + .as_array() + .into_iter() + .flatten() + .filter_map(|d| d["_docID"].as_str()) + .map(str::to_string) + .collect() +} + +/// The `name` of document `id` in a `list_users` reply; `None` when the +/// node does not have it. +pub(super) fn name_of(data: &Value, id: &str) -> Option { + data[COLLECTION] + .as_array()? + .iter() + .find(|d| d["_docID"] == id) + .map(|d| d["name"].as_str().unwrap_or_default().to_string()) +} + +/// Create `n` documents with `age` at `node`; their ids. +pub(super) async fn write_docs( + ch: &dyn Channel, + node: usize, + n: usize, + age: i64, +) -> Result> { + let data = ch.gql(node, create_users(n, age)).await?; + Ok(ids_in(&data, &format!("add_{COLLECTION}"))) +} + +pub(super) async fn doc_ids(ch: &dyn Channel, node: usize) -> Result> { + Ok(ids_in(&ch.gql(node, list_users()).await?, COLLECTION)) +} + +/// Poll `node` with `query` until `done` holds of the data or the settle +/// window passes; the data at the end. +pub(super) async fn settle( + ch: &dyn Channel, + node: usize, + query: &str, + done: impl Fn(&Value) -> bool, +) -> Result { + for poll in 0..SETTLE_POLLS { + let data = ch.gql(node, query.to_string()).await?; + if done(&data) || poll + 1 == SETTLE_POLLS { + return Ok(data); + } + tokio::time::sleep(Duration::from_millis(500)).await; + } + unreachable!("SETTLE_POLLS is positive") +} + +/// Poll `node` until it has every id in `want` or the settle window +/// passes; the ids it had at the end. +pub(super) async fn converge( + ch: &dyn Channel, + node: usize, + want: &[String], +) -> Result> { + let data = settle(ch, node, &list_users(), |data| { + let have = ids_in(data, COLLECTION); + want.iter().all(|w| have.contains(w)) + }) + .await?; + Ok(ids_in(&data, COLLECTION)) +} + +#[cfg(test)] +pub(super) mod fake { + use serde_json::json; + + use super::*; + + /// A data plane where a write at any node is on every node at once, + /// except node 0, which sees a document only if `sink_sees(age)`. + /// `add_` mints ids, `update_` renames, a query lists. + pub fn store( + sink_sees: impl Fn(i64) -> bool + 'static, + ) -> impl FnMut(usize, &str) -> Result { + let quoted = |rest: &str| rest.split('"').next().unwrap_or_default().to_string(); + let mut docs: Vec<(String, String, i64)> = Vec::new(); + move |node, q| { + if q.starts_with("mutation { add_") { + let ids: Vec = q + .split("name: \"") + .skip(1) + .map(|rest| { + let age = rest + .split("age: ") + .nth(1) + .unwrap() + .chars() + .take_while(char::is_ascii_digit) + .collect::() + .parse() + .unwrap(); + let id = format!("bae-{}", docs.len()); + docs.push((id.clone(), quoted(rest), age)); + json!({ "_docID": id }) + }) + .collect(); + return Ok(json!({ format!("add_{COLLECTION}"): ids })); + } + if q.starts_with("mutation { update_") { + let id = quoted(q.split("docID: \"").nth(1).unwrap()); + let name = quoted(q.split("name: \"").nth(1).unwrap()); + if let Some(doc) = docs.iter_mut().find(|d| d.0 == id) { + doc.1 = name; + } + return Ok(json!({})); + } + let seen: Vec = docs + .iter() + .filter(|(_, _, age)| node != 0 || sink_sees(*age)) + .map(|(id, name, _)| json!({ "_docID": id, "name": name })) + .collect(); + Ok(json!({ COLLECTION: seen })) + } + } +} + +#[cfg(test)] +mod tests { + use super::fake::store; + use super::*; + + #[test] + fn mutations_and_ids() { + let (a, b) = (create_users(2, 1), create_users(1, 1)); + assert!( + a.starts_with("mutation { add_User(input: [{name: \"u"), + "{a}" + ); + assert!(a.ends_with(", age: 1}]) { _docID } }"), "{a}"); + let name = |m: &str| { + m.split("name: ") + .nth(1) + .unwrap() + .split('"') + .nth(1) + .unwrap() + .to_string() + }; + assert_ne!(name(&a), name(&b), "names never repeat: {a} / {b}"); + assert_eq!( + update_user("bae-1", "x"), + "mutation { update_User(docID: \"bae-1\", input: {name: \"x\"}) { _docID } }" + ); + let mut s = store(|age| age == 1); + let made = s(1, &create_users(2, 1)).unwrap(); + assert_eq!(ids_in(&made, "add_User"), ["bae-0", "bae-1"]); + s(1, &create_users(1, 2)).unwrap(); + assert_eq!( + ids_in(&s(1, &list_users()).unwrap(), COLLECTION), + ["bae-0", "bae-1", "bae-2"] + ); + assert_eq!( + ids_in(&s(0, &list_users()).unwrap(), COLLECTION), + ["bae-0", "bae-1"] + ); + s(1, &update_user("bae-1", "x")).unwrap(); + let listed = s(0, &list_users()).unwrap(); + assert_eq!(name_of(&listed, "bae-1").as_deref(), Some("x")); + assert!(name_of(&listed, "bae-0").unwrap().starts_with('u')); + assert_eq!(name_of(&listed, "bae-2"), None); + } +} diff --git a/crates/soak/src/manage/hybrid.rs b/crates/soak/src/manage/hybrid.rs new file mode 100644 index 0000000..32d7778 --- /dev/null +++ b/crates/soak/src/manage/hybrid.rs @@ -0,0 +1,268 @@ +//! H1: Go nodes present as replication peers, never as manage targets (Go +//! has no manage protocol). The cases two Rust nodes can host run with the +//! Go nodes in the mesh, each its own row; H1's own row is about the Go +//! nodes: they converge on the source's documents after every writing +//! case, and their replicator sets are what the mesh gave them. + +use eyre::Result; +use serde_json::{json, Value}; + +use super::cases::{self, fail, skip, Channel}; +use super::data::{converge, doc_ids}; + +/// The rows H1 runs, in table order; `WRITERS` write at [`SOURCE`]. +const EMBEDDED: &str = "R2,A1,A2,S1,S3,S4"; +const WRITERS: &[&str] = &["S3", "S4"]; +const SOURCE: usize = 1; + +pub(super) async fn h1(ch: &mut dyn Channel) -> Result<()> { + let go = ch.go_nodes(); + if go.is_empty() { + return Err(skip("no Go nodes in the topology")); + } + let rust = ch.len(); + let before = go_replicator_sets(ch, &go)?; + let table = cases::all(); + // Embedded rows drain the channel's notes; H1's own wait until the end. + let mut notes = Vec::new(); + let mut verdict = Ok(()); + for case in cases::select(&table, Some(EMBEDDED))? { + let rows = cases::run_all(ch, &[case], rust).await; + ch.embed(rows); + if WRITERS.contains(&case.name) { + verdict = verdict.and(converged(ch, &go, case.name, &mut notes).await); + } + } + for note in notes { + ch.note(note); + } + verdict?; + let after = go_replicator_sets(ch, &go)?; + if before != after { + return Err(fail( + format!("the Go nodes' replicator sets untouched: {before}"), + after.to_string(), + )); + } + Ok(()) +} + +/// Every Go node has exactly the source's documents within the settle +/// window; the counts are noted either way. +async fn converged( + ch: &dyn Channel, + go: &[usize], + after: &str, + notes: &mut Vec, +) -> Result<()> { + let want = doc_ids(ch, SOURCE).await?; + let mut counts = Vec::new(); + let mut lagging = Vec::new(); + for &g in go { + let have = converge(ch, g, &want).await?; + counts.push(format!("go node {g} {}/{}", have.len(), want.len())); + if have.len() != want.len() || !want.iter().all(|w| have.contains(w)) { + lagging.push(counts.last().unwrap().clone()); + } + } + notes.push(format!("after {after}: {}", counts.join(", "))); + if lagging.is_empty() { + Ok(()) + } else { + Err(fail( + format!( + "every Go node with the source's {} documents after {after}", + want.len() + ), + lagging.join(", "), + )) + } +} + +/// Each Go node's replicators as (peer id, collection ids), the fields a +/// manage op could change; the status fields flip on their own. The list +/// is the Go CLI's `client.Replicator` shape. The mesh gave every Go node +/// one replicator per other node; another set is a finding in its own +/// right. +fn go_replicator_sets(ch: &dyn Channel, go: &[usize]) -> Result { + let mut sets = Vec::new(); + for &g in go { + let list = ch.replicators(g)?; + let mut set: Vec = list + .as_array() + .into_iter() + .flatten() + .map(|r| { + let mut cols = r["CollectionIDs"].as_array().cloned().unwrap_or_default(); + cols.sort_by_key(|v| v.to_string()); + json!([r["ID"], cols]) + }) + .collect(); + set.sort_by_key(|v| v.to_string()); + let mut want: Vec = (0..ch.len()) + .chain(go.iter().copied()) + .filter(|&n| n != g) + .map(|n| ch.peer_id(n)) + .collect(); + want.sort(); + let mut have: Vec = set + .iter() + .map(|r| r[0].as_str().unwrap_or_default().to_string()) + .collect(); + have.sort(); + if have != want { + return Err(fail( + format!("go node {g} with a replicator per mesh peer {want:?}"), + format!("{have:?} in {list}"), + )); + } + sets.push(json!({ "node": g, "replicators": set })); + } + Ok(Value::Array(sets)) +} + +#[cfg(test)] +mod tests { + use std::cell::RefCell; + + use super::super::cases::fake::*; + use super::super::cases::{run_all, CaseReport, Outcome}; + use super::super::data::fake::store; + use super::*; + + /// A Go node's list on the three-Rust, two-Go fake: one per other node. + fn go_mesh(node: usize) -> Value { + Value::Array( + (0..5) + .filter(|&p| p != node) + .map(|p| json!({"ID": format!("peer{p}"), "CollectionIDs": ["bafy-user"], "Status": (p == 0) as u8})) + .collect(), + ) + } + + fn hybrid(sees: impl Fn(usize) -> bool + 'static) -> Fake { + let mut fake = Fake::new(|_, _, _, _, op| admin_view(op)); + fake.go = vec![3, 4]; + fake.own = RefCell::new(Box::new(|node| Ok(go_mesh(node)))); + let mut inner = store(|age| age == 1); + fake.gql = RefCell::new(Box::new(move |node, q| { + if !sees(node) && !q.starts_with("mutation") { + return Ok(json!({ "User": [] })); + } + inner(node, q) + })); + fake + } + + /// H1's rows: the embedded ones, then its own. + async fn run_h1(mut fake: Fake) -> Vec { + run_all(&mut fake, &[&by_name("H1")], 3).await + } + + #[tokio::test] + async fn h1_skips_without_go_nodes() { + let rows = run_h1(Fake::new(|_, _, _, _, op| admin_view(op))).await; + assert_eq!(rows.len(), 1); + assert_eq!( + rows[0].outcome, + Outcome::Skip { + reason: "no Go nodes in the topology".into() + } + ); + } + + #[tokio::test] + async fn h1_embeds_the_two_node_cases_and_notes_the_go_counts_after_each_writer() { + let rows = run_h1(hybrid(|_| true)).await; + let names: Vec<_> = rows.iter().map(|r| r.name).collect(); + assert_eq!(names, ["R2", "A1", "A2", "S1", "S3", "S4", "H1"]); + let h1 = rows.last().unwrap(); + assert_eq!(h1.outcome, Outcome::Pass, "{:?}", h1.notes); + assert_eq!( + h1.notes, + [ + "after S3: go node 3 4/4, go node 4 4/4", + "after S4: go node 3 5/5, go node 4 5/5" + ] + ); + assert!( + rows[5].notes.is_empty(), + "S4 kept H1's note: {:?}", + rows[5].notes + ); + } + + #[tokio::test(start_paused = true)] + async fn h1_fails_when_a_go_node_lags_after_the_settle_window() { + let rows = run_h1(hybrid(|node| node != 4)).await; + assert_eq!(rows.len(), 7, "every row still runs"); + let h1 = rows.last().unwrap(); + assert!( + matches!(&h1.outcome, Outcome::Fail { expected, got } if expected == "every Go node with the source's 4 documents after S3" && got == "go node 4 0/4"), + "{:?}", + h1.outcome + ); + assert_eq!( + h1.notes, + [ + "after S3: go node 3 4/4, go node 4 0/4", + "after S4: go node 3 5/5, go node 4 0/5" + ] + ); + } + + #[tokio::test] + async fn h1_fails_when_a_go_replicator_set_changed_and_ignores_status_flips() { + let mut fake = hybrid(|_| true); + let reads = std::cell::Cell::new(0); + fake.own = RefCell::new(Box::new(move |node| { + reads.set(reads.get() + 1); + let mut list = go_mesh(node); + list[0]["Status"] = json!(reads.get()); + if node == 4 && reads.get() > 2 { + list[0]["CollectionIDs"] = json!(["bafy-other"]); + } + Ok(list) + })); + let rows = run_h1(fake).await; + let outcome = &rows.last().unwrap().outcome; + assert!( + matches!(outcome, Outcome::Fail { expected, got } if expected.starts_with("the Go nodes' replicator sets untouched") && got.contains(r#"{"node":4,"replicators":[["peer0",["bafy-other"]],["peer1",["bafy-user"]]"#)), + "{outcome:?}" + ); + } + + #[tokio::test] + async fn h1_fails_when_a_go_node_does_not_show_the_mesh() { + let mut fake = hybrid(|_| true); + fake.own = RefCell::new(Box::new(|node| { + Ok(if node == 3 { json!([]) } else { go_mesh(node) }) + })); + let rows = run_h1(fake).await; + assert_eq!( + rows.last().unwrap().outcome, + Outcome::Fail { + expected: r#"go node 3 with a replicator per mesh peer ["peer0", "peer1", "peer2", "peer4"]"#.into(), + got: "[] in []".into() + } + ); + } + + #[tokio::test] + async fn h1_fails_when_a_go_replicator_is_not_a_mesh_peer() { + let mut fake = hybrid(|_| true); + fake.own = RefCell::new(Box::new(|node| { + let mut list = go_mesh(node); + if node == 4 { + list[3]["ID"] = json!("peer9"); + } + Ok(list) + })); + let rows = run_h1(fake).await; + let outcome = &rows.last().unwrap().outcome; + assert!( + matches!(outcome, Outcome::Fail { expected, got } if expected.starts_with("go node 4 with a replicator per mesh peer") && got.starts_with(r#"["peer0", "peer1", "peer2", "peer9"] in "#)), + "{outcome:?}" + ); + } +} diff --git a/crates/soak/src/manage/mod.rs b/crates/soak/src/manage/mod.rs new file mode 100644 index 0000000..290d6b2 --- /dev/null +++ b/crates/soak/src/manage/mod.rs @@ -0,0 +1,471 @@ +//! `soak manage`: pass/fail cases on the P2P management channel, on a +//! NAC-enabled Rust mesh. The caller POSTs to a relay node's HTTP API and +//! that node relays a signed request over P2P to the target. + +pub mod actors; +pub mod authz; +pub mod bounds; +pub mod cases; +pub mod client; +pub mod data; +pub mod hybrid; +pub mod partition; +pub mod report; +pub mod routing; +pub mod state; + +use std::cell::RefCell; +use std::path::Path; + +use eyre::{ensure, eyre, Result, WrapErr}; +use futures::future::{BoxFuture, LocalBoxFuture}; +use serde_json::{json, Value}; + +use crate::auth::auth_token; +use crate::nodes::{NodeKind, Nodes}; +use crate::{flag, has_flag, start_nodes, RunArgs, Topology, Transport}; +use actors::{Actor, Actors}; +use cases::{CaseReport, Channel, OpRecord, Verb}; + +/// `age` is immutable so a replication filter may use it (S3). Go has no +/// `@immutable`; the directive does not enter the collection id, so a Go +/// node takes the plain form and shares the collection. +const SCHEMA: &str = "type User { name: String age: Int @immutable }"; +const GO_SCHEMA: &str = "type User { name: String age: Int }"; + +/// Past this many doc refs a mutate's after-state is not report material +/// (the bounds cases send hundreds of thousands). +const STATE_READ_MAX_DOCS: usize = 100; + +pub async fn run(out: &Path, a: RunArgs) -> Result<()> { + let topology = a + .topology + .ok_or_else(|| eyre!("manage needs --topology rg"))?; + accept(topology, a.transport)?; + ensure!( + !a.docker && !has_flag("docker"), + "manage --docker: unsupported yet" + ); + let table = cases::all(); + let mut selected = cases::select(&table, flag("cases").as_deref())?; + if has_flag("locate-size-bound") && !selected.iter().any(|c| c.name == "B3") { + selected.extend(table.iter().filter(|c| c.name == "B3")); + } + let mut nodes = start_nodes(out, &a).await?; + let result = drive(out, topology, a.transport, &mut nodes, &selected).await; + let shutdown = nodes.shutdown().await; + result?; + shutdown +} + +/// Every manage target is Rust; Go nodes are replication peers only, and +/// only over libp2p, the one transport Go speaks. +fn accept(topology: Topology, transport: Transport) -> Result<()> { + ensure!(topology.rust >= 2, "manage needs at least two Rust nodes"); + ensure!( + topology.go == 0 || transport == Transport::Libp2p, + "manage --transport iroh: Go nodes cannot speak iroh; use --topology {}r0g", + topology.rust + ); + Ok(()) +} + +async fn drive( + out: &Path, + topology: Topology, + transport: Transport, + nodes: &mut Nodes, + selected: &[&cases::Case], +) -> Result<()> { + let owner = match &*nodes { + Nodes::Process { cluster, .. } => cluster.startup_identity(), + Nodes::Docker(_) => None, + } + .ok_or_else(|| eyre!("NAC cluster has no startup identity"))? + .to_string(); + let n = nodes.len(); + let mut addrs = Vec::new(); + for i in 0..n { + let info = nodes + .client(i) + .p2p_info_with_identity(&owner) + .wrap_err_with(|| format!("p2p info on {}", nodes.name(i)))?; + let addr = info[0] + .as_str() + .ok_or_else(|| eyre!("{} has no P2P address: {info}", nodes.name(i)))?; + addrs.push(addr.to_string()); + } + let peer_ids: Vec = addrs.iter().map(|a| peer_id_of(a).to_string()).collect(); + for (i, pid) in peer_ids.iter().enumerate() { + println!("{} at {} peer id {pid}", nodes.name(i), nodes.api_url(i)); + } + wire_mesh(nodes, &owner, &addrs)?; + let actors = Actors::generate(&nodes.binaries()?[0])?; + for i in 0..topology.rust { + actors.grant_on(nodes, i, &owner)?; + } + println!( + "actors: admin {} operator {} ({}) outsider {}", + actors.admin.did, + actors.operator.did, + actors::OPERATOR_GRANTS.join(","), + actors.outsider.did + ); + let manifest = json!({ + "arm": "manage", + "topology": topology.label(), + "transport": transport.label(), + "nodes": (0..n).map(|i| json!({ + "name": nodes.name(i), "runtime": nodes.kind(i), "api_url": nodes.api_url(i), + "p2p_addr": addrs[i], "peer_id": peer_ids[i], + })).collect::>(), + "owner_key_hex": owner, + "actors": actors, + "rust_binary": std::env::var("DEFRA_RUST_BINARY").unwrap_or_default(), + "cases": selected.iter().map(|c| c.name).collect::>(), + }); + std::fs::write( + out.join("manifest.json"), + serde_json::to_string_pretty(&manifest)?, + )?; + let mut live = Live { + http: crate::executor::http_client(std::time::Duration::from_secs(60)), + urls: (0..n).map(|i| nodes.api_url(i)).collect(), + nodes, + rust: topology.rust, + transport, + addrs, + peer_ids, + courier: owner, + actors, + records: RefCell::default(), + notes: Vec::new(), + embedded: Vec::new(), + }; + let reports = cases::run_all(&mut live, selected, topology.rust).await; + report::write(out, &topology.label(), transport.label(), &reports)?; + println!("report: {}", out.join("cases.md").display()); + Ok(()) +} + +/// The peer id `/p2p/info` reports: the multiaddr's last `/p2p/` segment +/// under libp2p, the endpoint id after `host:port/p2p/` under iroh. +fn peer_id_of(addr: &str) -> &str { + addr.rsplit("/p2p/").next().unwrap_or(addr) +} + +/// Schema, peer connections and a replicator per ordered pair, Go nodes +/// included, every verb as the NAC owner. No collection subscribe: +/// replicators are the only delivery path, `CollectionAdd` stays a clean +/// toggle for the cases, and a Go node forwards only what a replicator +/// gave it. +fn wire_mesh(nodes: &Nodes, owner: &str, addrs: &[String]) -> Result<()> { + let n = nodes.len(); + for i in 0..n { + let name = nodes.name(i); + let client = nodes.client(i); + let schema = match nodes.kind(i) { + NodeKind::Rust => SCHEMA, + NodeKind::Go => GO_SCHEMA, + }; + client + .schema_add_with_identity(schema, owner) + .wrap_err_with(|| format!("schema on {name}"))?; + let others: Vec<&str> = (0..n) + .filter(|j| *j != i) + .map(|j| addrs[j].as_str()) + .collect(); + client + .p2p_connect_with_identity(&others, owner) + .wrap_err_with(|| format!("connect from {name}"))?; + for j in &others { + client + .p2p_replicator_set_with_identity(&[cases::COLLECTION], j, owner) + .wrap_err_with(|| format!("replicator {name} -> {j}"))?; + } + } + Ok(()) +} + +/// The real channel: actors' tokens, the relay's HTTP API, one record per +/// request. After a mutate it reads the target's list for that op's +/// family as admin, so the report shows the state every op left behind. +/// Verbs go to the harness, except the NAC toggle, which is the node's own +/// `/acp/node/{disable,re-enable}` route as the owner. +struct Live<'n> { + nodes: &'n mut Nodes, + /// Nodes `0..rust` are Rust, the rest Go. + rust: usize, + transport: Transport, + http: reqwest::Client, + urls: Vec, + addrs: Vec, + peer_ids: Vec, + courier: String, + actors: Actors, + records: RefCell>, + notes: Vec, + embedded: Vec, +} + +fn family_list(kind: &str) -> Option<&'static str> { + ["Replicator", "Collection", "Document"] + .into_iter() + .find(|f| kind.starts_with(f)) + .map(|f| match f { + "Replicator" => "ReplicatorList", + "Collection" => "CollectionList", + _ => "DocumentList", + }) +} + +impl Live<'_> { + fn pid(&self, node: usize) -> Result { + self.nodes + .pid(node) + .ok_or_else(|| eyre!("{}: no process to signal", self.nodes.name(node))) + } + + /// POST to `node`'s own `/api/v0/{route}` as the owner. + async fn post_own(&self, node: usize, route: &str, body: Option<&Value>) -> Result<()> { + let url = &self.urls[node]; + let mut req = self + .http + .post(format!("{url}/api/v0/{route}")) + .bearer_auth(auth_token(&self.courier, url)?); + if let Some(body) = body { + req = req.json(body); + } + let resp = req + .send() + .await + .wrap_err_with(|| format!("{route} at {url}"))?; + let status = resp.status(); + ensure!( + status.is_success(), + "{route} at {url}: {status} {}", + resp.text().await.unwrap_or_default() + ); + Ok(()) + } + + async fn post( + &self, + relay: usize, + target: usize, + audience: usize, + actor: Actor, + op: &Value, + ) -> Result { + let token = self.actors.token(actor, &self.peer_ids[audience])?; + client::post( + &self.http, + &self.urls[relay], + &self.courier, + &self.addrs[target], + &token, + op, + ) + .await + } +} + +impl Channel for Live<'_> { + fn len(&self) -> usize { + self.rust + } + + fn go_nodes(&self) -> Vec { + (self.rust..self.urls.len()).collect() + } + + fn transport(&self) -> Transport { + self.transport + } + + fn addr(&self, node: usize) -> String { + self.addrs[node].clone() + } + + fn peer_id(&self, node: usize) -> String { + self.peer_ids[node].clone() + } + + fn send_for<'a>( + &'a self, + relay: usize, + target: usize, + audience: usize, + actor: Actor, + op: Value, + ) -> LocalBoxFuture<'a, Result> { + Box::pin(async move { + let kind = op["Kind"].as_str().unwrap_or_default().to_string(); + let reply = self.post(relay, target, audience, actor, &op).await?; + let small = op["docs"].as_array().map_or(0, Vec::len) <= STATE_READ_MAX_DOCS; + let target_state = match family_list(&kind).filter(|_| !client::is_query(&op) && small) + { + Some(list) => Some( + match self + .post( + relay, + target, + target, + Actor::Admin, + &json!({ "Kind": list }), + ) + .await + { + Ok(r) => r.body, + Err(e) => json!({ "error": format!("{e:#}") }), + }, + ), + None => None, + }; + self.records.borrow_mut().push(OpRecord { + relay, + target, + actor, + kind, + status: reply.status, + latency_ms: reply.latency_ms, + target_state, + }); + Ok(reply) + }) + } + + fn control<'a>(&'a mut self, verb: Verb) -> LocalBoxFuture<'a, Result<()>> { + Box::pin(async move { + match verb { + Verb::Stop(i) => self.nodes.stop(i).await, + Verb::Start(i) => self.nodes.start_stopped(i).await, + Verb::Regrant(i) => self.actors.grant_on(self.nodes, i, &self.courier), + Verb::Grant { + node, + actor, + relation, + } => self + .nodes + .client(node) + .acp_node_relationship_add( + relation, + &self.actors.identity(actor).did, + &self.courier, + ) + .map(drop), + Verb::Revoke { + node, + actor, + relation, + } => self + .nodes + .client(node) + .acp_node_relationship_delete( + relation, + &self.actors.identity(actor).did, + &self.courier, + ) + .map(drop), + Verb::Nac { node, on } => { + let route = if on { "re-enable" } else { "disable" }; + self.post_own(node, &format!("acp/node/{route}"), None) + .await + } + Verb::Partition(i) => self.nodes.partition(i).await, + Verb::Rejoin(i) => self.nodes.rejoin(i).await, + Verb::PauseAfter { node, delay_ms } => { + let pid = self.pid(node)?; + tokio::spawn(async move { + tokio::time::sleep(std::time::Duration::from_millis(delay_ms)).await; + if let Err(e) = crate::nodes::signal(pid, "-STOP") { + eprintln!("pause of pid {pid}: {e:#}"); + } + }); + Ok(()) + } + Verb::Resume(i) => crate::nodes::signal(self.pid(i)?, "-CONT"), + Verb::LocalReplicatorAdd { + node, + peer, + filters, + } => { + let body = json!({ + "Collections": [cases::COLLECTION], + "Addresses": [self.addrs[peer]], + "Filters": filters, + }); + self.post_own(node, "p2p/replicators", Some(&body)).await + } + } + }) + } + + fn gql(&self, node: usize, query: String) -> BoxFuture<'static, Result> { + let http = self.http.clone(); + let url = self.urls[node].clone(); + let token = auth_token(&self.courier, &url); + Box::pin(async move { + crate::executor::gql_as(&http, &url, &query, Some(&token?)) + .await + .map_err(|e| eyre!("{e} at {url}")) + }) + } + + fn replicators(&self, node: usize) -> Result { + self.nodes + .client(node) + .p2p_replicator_list_with_identity(&self.courier) + .wrap_err_with(|| format!("replicator list on {}", self.nodes.name(node))) + } + + fn can_partition(&self) -> bool { + self.nodes.supports_partition() + } + + fn note(&mut self, text: String) { + self.notes.push(text); + } + + fn take_records(&mut self) -> Vec { + self.records.take() + } + + fn take_notes(&mut self) -> Vec { + std::mem::take(&mut self.notes) + } + + fn embed(&mut self, reports: Vec) { + self.embedded.extend(reports); + } + + fn take_embedded(&mut self) -> Vec { + std::mem::take(&mut self.embedded) + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn go_nodes_join_on_libp2p_only() { + let t = |rust, go| Topology { rust, go }; + assert!(accept(t(2, 0), Transport::Libp2p).is_ok()); + assert!(accept(t(2, 2), Transport::Libp2p).is_ok()); + assert!(accept(t(3, 0), Transport::Iroh).is_ok()); + let refused = accept(t(2, 2), Transport::Iroh).unwrap_err().to_string(); + assert!(refused.contains("Go nodes cannot speak iroh"), "{refused}"); + assert!(accept(t(1, 2), Transport::Libp2p).is_err()); + } + + #[test] + fn peer_id_of_reads_the_libp2p_and_iroh_address_forms() { + assert_eq!( + peer_id_of("/ip4/127.0.0.1/tcp/9171/p2p/12D3KooWabc"), + "12D3KooWabc" + ); + assert_eq!(peer_id_of("127.0.0.1:9171/p2p/1a2b3c"), "1a2b3c"); + assert_eq!(peer_id_of("1a2b3c"), "1a2b3c"); + } +} diff --git a/crates/soak/src/manage/partition.rs b/crates/soak/src/manage/partition.rs new file mode 100644 index 0000000..1421a5e --- /dev/null +++ b/crates/soak/src/manage/partition.rs @@ -0,0 +1,369 @@ +//! Partition and concurrency cases: an op into a cut-off target, and ops +//! beside a write burst. + +use eyre::Result; +use serde_json::json; +use tokio::time::Instant; + +use super::actors::Actor; +use super::cases::{ + collection_add, collection_remove, expect_status, fail, managed_state, skip, Channel, Verb, + COLLECTION, +}; +use super::data::{converge, create_users, doc_ids, ids_in}; +use super::state::{document_add, document_remove}; + +/// Documents the C1 burst writes at the source. +const BURST: usize = 50; + +/// A well-formed id no document has; C1 tracks and untracks it. +const SYNTH_DOC: &str = "bae-00000000-0000-0000-0000-0000000000c1"; + +/// P1: `CollectionAdd` while the target is cut off, then healed. The +/// outcome is recorded, not judged: the relay must answer, whatever it +/// says, and after the rejoin the op is on the target or it is not. +pub(super) async fn p1(ch: &mut dyn Channel) -> Result<()> { + if !ch.can_partition() { + return Err(skip( + "this backend cannot partition a node (process nodes; manage has no --docker yet)", + )); + } + let (relay, target) = (0, 1); + ch.control(Verb::Partition(target)).await?; + let probe = ch.send(relay, target, Actor::Admin, collection_add()).await; + ch.control(Verb::Rejoin(target)).await?; + let probe = probe.map_err(|e| { + fail( + "a reply while the target is partitioned", + format!("no reply: {e:#}"), + ) + })?; + let list = ch + .send( + relay, + target, + Actor::Admin, + json!({ "Kind": "CollectionList" }), + ) + .await?; + expect_status(&list, 200, "CollectionList after the rejoin")?; + let landed = list.body["values"] + .as_array() + .is_some_and(|v| v.iter().any(|c| c == COLLECTION)); + if landed { + let r = ch + .send(relay, target, Actor::Admin, collection_remove()) + .await?; + expect_status(&r, 200, "CollectionRemove (restore)")?; + } + ch.note(format!( + "CollectionAdd while partitioned: {} after {} ms: {}; after rejoin: {}", + probe.status, + probe.latency_ms, + probe.body, + if landed { "landed" } else { "lost" } + )); + Ok(()) +} + +/// C1: a fixed op sequence on the source while a burst of documents is +/// written there: every op lands, at least one is answered before the +/// burst's own reply comes back (the timestamps are noted), the source's +/// managed state is back where it started, and every burst document +/// reaches the sink. The sink's total against the source's is noted, not +/// asserted: documents earlier cases left in the mesh are not C1's. +pub(super) async fn c1(ch: &mut dyn Channel) -> Result<()> { + let (relay, source) = (0, 1); + let sink = relay; + let before = managed_state(ch, relay, source).await?; + let started = Instant::now(); + let write = ch.gql(source, create_users(BURST, 1)); + let burst = tokio::spawn(async move { (write.await, Instant::now()) }); + let ops = [ + collection_add(), + document_add(SYNTH_DOC), + collection_remove(), + document_remove(SYNTH_DOC), + ]; + let mut spans = Vec::new(); + for op in ops { + let kind = op["Kind"].as_str().unwrap_or_default().to_string(); + let sent = started.elapsed(); + let r = ch.send(relay, source, Actor::Admin, op).await?; + spans.push((kind.clone(), sent, started.elapsed())); + expect_status(&r, 200, &format!("{kind} during the burst"))?; + } + let (data, ended) = burst.await?; + let ended = ended - started; + let written = ids_in(&data?, &format!("add_{COLLECTION}")); + let inside = spans + .iter() + .filter(|(_, _, replied)| *replied < ended) + .count(); + ch.note(format!( + "burst 0..{} ms; {}; {inside} of {} ops inside", + ended.as_millis(), + spans + .iter() + .map(|(kind, sent, replied)| format!( + "{kind} {}..{} ms", + sent.as_millis(), + replied.as_millis() + )) + .collect::>() + .join(", "), + spans.len() + )); + if inside == 0 { + return Err(fail( + "an op answered while the burst was in flight", + format!( + "sequential, not concurrent: the burst ended at {} ms, the first op was answered at {} ms", + ended.as_millis(), + spans[0].2.as_millis() + ), + )); + } + if written.len() != BURST { + return Err(fail( + format!("{BURST} documents from the burst"), + format!("{}", written.len()), + )); + } + let have = converge(ch, sink, &written).await?; + let after = managed_state(ch, relay, source).await?; + if after != before { + return Err(fail( + format!("source state back where it started: {before}"), + after.to_string(), + )); + } + let at_source = doc_ids(ch, source).await?.len(); + ch.note(format!( + "after settle: sink {} of the source's {at_source} documents", + have.len() + )); + let missing: Vec<&str> = written + .iter() + .filter(|w| !have.contains(w)) + .map(String::as_str) + .collect(); + if !missing.is_empty() { + return Err(fail( + format!("the sink with all {BURST} burst documents after the settle window"), + format!( + "{} of {BURST}; missing {}", + BURST - missing.len(), + missing.join(", ") + ), + )); + } + Ok(()) +} + +#[cfg(test)] +mod tests { + use std::cell::RefCell; + use std::rc::Rc; + + use super::super::cases::fake::*; + use super::super::cases::Outcome; + use super::super::client::Reply; + use super::super::data::fake::store; + use super::*; + + fn replied(status: u16, ms: u64, body: &str) -> Result { + Ok(Reply { + status, + body: json!({ "error": body }), + latency_ms: ms, + }) + } + + /// Node 1 answers `cut` while partitioned; after the rejoin its + /// collection list has the collection iff `landed`. + fn partitioned(cut: Result, landed: bool) -> Fake { + let verbs: Rc>> = Default::default(); + let seen = verbs.clone(); + let cut = std::cell::Cell::new(Some(cut)); + let mut fake = Fake::new(move |_, target, _, _, op| { + if target == 1 && seen.borrow().last() == Some(&Verb::Partition(1)) { + return cut.take().expect("one op while partitioned"); + } + if op["Kind"] == "CollectionList" { + let values: Vec<&str> = if landed { vec![COLLECTION] } else { vec![] }; + return ok(json!({"Kind": "Strings", "values": values})); + } + admin_view(op) + }); + fake.verbs = verbs; + fake + } + + #[tokio::test] + async fn p1_records_lost_or_landed_and_never_fails_a_recorded_outcome() { + let (outcome, notes) = run_noted( + "P1", + partitioned(replied(400, 10_000, "dial timeout"), false), + ) + .await; + assert_eq!(outcome, Outcome::Pass); + assert_eq!( + notes, + ["CollectionAdd while partitioned: 400 after 10000 ms: {\"error\":\"dial timeout\"}; after rejoin: lost"] + ); + + let (tx, rx) = std::sync::mpsc::channel(); + let mut fake = partitioned(replied(200, 15, "ok"), true); + let inner = RefCell::new(fake.rule.replace(Box::new(|_, _, _, _, _| status(200)))); + fake.rule = RefCell::new(Box::new(move |r, t, a, actor, op| { + tx.send(op["Kind"].as_str().unwrap().to_string()).unwrap(); + (inner.borrow_mut())(r, t, a, actor, op) + })); + let (outcome, notes) = run_noted("P1", fake).await; + assert_eq!(outcome, Outcome::Pass); + assert!(notes[0].ends_with("after rejoin: landed"), "{notes:?}"); + assert!( + rx.try_iter().any(|k| k == "CollectionRemove"), + "a landed op is restored" + ); + } + + #[tokio::test] + async fn p1_fails_only_on_a_hang_and_still_rejoins() { + let (outcome, verbs) = run_fake( + "P1", + partitioned(Err(eyre::eyre!("operation timed out")), false), + ) + .await; + assert!( + matches!(&outcome, Outcome::Fail { got, .. } if got.contains("timed out")), + "{outcome:?}" + ); + assert_eq!(verbs, [Verb::Partition(1), Verb::Rejoin(1)]); + } + + #[tokio::test] + async fn p1_skips_where_nothing_can_partition() { + let mut fake = Fake::new(|_, _, _, _, op| admin_view(op)); + fake.partition = false; + let (outcome, verbs) = run_fake("P1", fake).await; + assert!( + matches!(&outcome, Outcome::Skip { reason } if reason.contains("cannot partition")), + "{outcome:?}" + ); + assert!(verbs.is_empty()); + } + + /// `fake` with a data plane whose sink sees `sink_sees`, answering + /// 100 ms later, so a burst is in flight while the ops go out. + fn bursting(mut fake: Fake, sink_sees: impl Fn(i64) -> bool + 'static) -> Fake { + fake.gql = RefCell::new(Box::new(store(sink_sees))); + fake.gql_ms = 100; + fake + } + + #[tokio::test(start_paused = true)] + async fn c1_lands_every_op_inside_the_burst_and_wants_the_sink_to_catch_up() { + let (tx, rx) = std::sync::mpsc::channel(); + let fake = Fake::new(move |_, target, _, _, op| { + tx.send((target, op["Kind"].as_str().unwrap().to_string())) + .unwrap(); + admin_view(op) + }); + let (outcome, notes) = run_noted("C1", bursting(fake, |_| true)).await; + assert_eq!(outcome, Outcome::Pass); + assert_eq!( + notes, + [ + "burst 0..100 ms; CollectionAdd 0..0 ms, DocumentAdd 0..0 ms, CollectionRemove 0..0 ms, DocumentRemove 0..0 ms; 4 of 4 ops inside", + "after settle: sink 50 of the source's 50 documents" + ] + ); + let kinds: Vec = rx + .try_iter() + .filter(|(t, k)| *t == 1 && !k.ends_with("List")) + .map(|(_, k)| k) + .collect(); + assert_eq!( + kinds, + [ + "CollectionAdd", + "DocumentAdd", + "CollectionRemove", + "DocumentRemove" + ] + ); + } + + #[tokio::test(start_paused = true)] + async fn c1_fails_when_no_op_is_answered_while_the_burst_is_in_flight() { + let mut fake = Fake::new(|_, _, _, _, op| admin_view(op)); + fake.gql = RefCell::new(Box::new(store(|_| true))); + // The burst answers at once, so it is over before the first op replies. + let (outcome, notes) = run_noted("C1", fake).await; + assert!( + matches!(&outcome, Outcome::Fail { expected, got } if expected.contains("in flight") && got.starts_with("sequential, not concurrent")), + "{outcome:?}" + ); + assert_eq!( + notes, + ["burst 0..0 ms; CollectionAdd 0..0 ms, DocumentAdd 0..0 ms, CollectionRemove 0..0 ms, DocumentRemove 0..0 ms; 0 of 4 ops inside"] + ); + } + + #[tokio::test(start_paused = true)] + async fn c1_fails_when_a_burst_document_never_reaches_the_sink() { + let fake = Fake::new(|_, _, _, _, op| admin_view(op)); + let (outcome, _) = run_fake("C1", bursting(fake, |_| false)).await; + assert!( + matches!(&outcome, Outcome::Fail { expected, got } if expected == "the sink with all 50 burst documents after the settle window" && got.starts_with("0 of 50; missing bae-0, bae-1, ")), + "{outcome:?}" + ); + } + + #[tokio::test(start_paused = true)] + async fn c1_notes_a_leftover_count_gap_without_failing() { + let mut fake = Fake::new(|_, _, _, _, op| admin_view(op)); + let mut inner = store(|age| age == 1); + inner(1, &create_users(1, 2)).unwrap(); + fake.gql = RefCell::new(Box::new(inner)); + fake.gql_ms = 100; + let (outcome, notes) = run_noted("C1", fake).await; + assert_eq!(outcome, Outcome::Pass, "{notes:?}"); + assert_eq!( + notes[1], + "after settle: sink 50 of the source's 51 documents" + ); + } + + #[tokio::test(start_paused = true)] + async fn c1_fails_on_a_refused_op_or_drifted_state() { + let refused = Fake::new(|_, _, _, _, op| { + if op["Kind"] == "DocumentAdd" { + status(400) + } else { + admin_view(op) + } + }); + let (outcome, _) = run_fake("C1", bursting(refused, |_| true)).await; + assert!( + matches!(&outcome, Outcome::Fail { expected, .. } if expected.contains("DocumentAdd during the burst")), + "{outcome:?}" + ); + + let mut removed = false; + let drifted = Fake::new(move |_, _, _, _, op| { + removed |= op["Kind"] == "CollectionRemove"; + if op["Kind"] == "CollectionList" && removed { + return ok(json!({"Kind": "Strings", "values": []})); + } + admin_view(op) + }); + let (outcome, _) = run_fake("C1", bursting(drifted, |_| true)).await; + assert!( + matches!(&outcome, Outcome::Fail { expected, .. } if expected.contains("back where it started")), + "{outcome:?}" + ); + } +} diff --git a/crates/soak/src/manage/report.rs b/crates/soak/src/manage/report.rs new file mode 100644 index 0000000..0f9427d --- /dev/null +++ b/crates/soak/src/manage/report.rs @@ -0,0 +1,70 @@ +//! `summary.json` and `cases.md` under `--out`, beside `manifest.json`. + +use std::path::Path; + +use eyre::Result; +use serde_json::json; + +use super::cases::{CaseReport, Outcome}; + +pub fn write(out: &Path, topology: &str, transport: &str, reports: &[CaseReport]) -> Result<()> { + let summary = json!({ "topology": topology, "transport": transport, "cases": reports }); + std::fs::write( + out.join("summary.json"), + serde_json::to_string_pretty(&summary)?, + )?; + std::fs::write(out.join("cases.md"), markdown(topology, transport, reports))?; + Ok(()) +} + +fn markdown(topology: &str, transport: &str, reports: &[CaseReport]) -> String { + let mut md = format!("# soak manage ({topology}, {transport})\n\n| case | outcome | ops | slowest ms | detail |\n|---|---|---|---|---|\n"); + for r in reports { + let (outcome, detail) = match &r.outcome { + Outcome::Pass => ("pass", String::new()), + Outcome::Fail { expected, got } => ("FAIL", format!("expected {expected}; got {got}")), + Outcome::Skip { reason } => ("skip", reason.clone()), + Outcome::Infra { error } => ("INFRA", error.clone()), + }; + let detail: Vec<&str> = std::iter::once(detail.as_str()) + .chain(r.notes.iter().map(String::as_str)) + .filter(|s| !s.is_empty()) + .collect(); + let slowest = r.ops.iter().map(|o| o.latency_ms).max().unwrap_or(0); + md.push_str(&format!( + "| {} | {outcome} | {} | {slowest} | {} |\n", + r.name, + r.ops.len(), + detail.join("; ").replace('|', "\\|").replace('\n', " ") + )); + } + md +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn markdown_names_the_topology_and_transport() { + assert!(markdown("3r0g", "iroh", &[]).starts_with("# soak manage (3r0g, iroh)\n")); + } + + #[test] + fn markdown_keeps_the_notes_beside_a_fail() { + let failed = CaseReport { + name: "R3", + outcome: Outcome::Fail { + expected: "200".into(), + got: "403".into(), + }, + ops: vec![], + notes: vec!["stopped target: pass".into()], + }; + let md = markdown("3r0g", "libp2p", &[failed]); + assert!( + md.ends_with("| R3 | FAIL | 0 | 0 | expected 200; got 403; stopped target: pass |\n"), + "{md}" + ); + } +} diff --git a/crates/soak/src/manage/routing.rs b/crates/soak/src/manage/routing.rs new file mode 100644 index 0000000..22930d4 --- /dev/null +++ b/crates/soak/src/manage/routing.rs @@ -0,0 +1,610 @@ +//! Routing cases: the relay reaches the target whatever their relationship. + +use std::time::Duration; + +use eyre::{Result, WrapErr}; +use serde_json::json; + +use super::actors::Actor; +use super::cases::{ + expect_status, fail, replicator_add, replicator_delete, replicators_for, Channel, Verb, + COLLECTION, +}; +use super::client::Reply; +use crate::Transport; + +/// The relay gives a libp2p dial 10 s; a clean error later than this is a +/// hang. +const LIBP2P_DIAL_BUDGET_MS: u64 = 15_000; + +/// Iroh gives up on a dead peer only after ~30 s (defradb.rs +/// `endpoint_commands.rs:761-766`), so its budget admits that. +const IROH_DIAL_BUDGET_MS: u64 = 35_000; + +/// A restarted target gets this long to answer its own HTTP as the owner +/// before the restart half runs; longer is a harness fault, not a finding. +const READY_BUDGET: Duration = Duration::from_secs(30); + +fn dial_budget_ms(t: Transport) -> u64 { + match t { + Transport::Libp2p => LIBP2P_DIAL_BUDGET_MS, + Transport::Iroh => IROH_DIAL_BUDGET_MS, + } +} + +/// R1: for every ordered (relay, target, source) of distinct nodes, admin +/// drops and restores the target's replicator to the source through the +/// relay, and the target's list follows each op. +pub(super) async fn r1(ch: &mut dyn Channel) -> Result<()> { + let n = ch.len(); + for relay in 0..n { + for target in (0..n).filter(|t| *t != relay) { + for source in (0..n).filter(|s| *s != relay && *s != target) { + let addr = ch.addr(source); + let peer = ch.peer_id(source); + let steps = [ + (replicator_delete(&addr), 0, "ReplicatorDelete"), + (replicator_add(&addr), 1, "ReplicatorAdd"), + ]; + for (op, want, name) in steps { + let what = format!("{name} via {relay} on {target} for {source}"); + let r = ch.send(relay, target, Actor::Admin, op).await?; + expect_status(&r, 200, &what)?; + let list = ch + .send( + relay, + target, + Actor::Admin, + json!({ "Kind": "ReplicatorList" }), + ) + .await?; + let got = replicators_for(&list.body, &peer); + if got != want { + return Err(fail( + format!("{want} replicator entry for node {source} after {what}"), + format!("{got} in {}", list.body), + )); + } + } + } + } + } + Ok(()) +} + +/// R3, two halves, each noted: the target (node 1) is stopped before the +/// call and the relay answers a clean 400 within the transport's dial +/// budget; then, once the target is back and answers its own HTTP, the +/// relay serves it again. A missing reply is the hang the first half +/// exists to catch, not a harness fault. The outcome is the first half +/// that fails. The grants are re-applied after the check: a node that +/// comes back without them is reported here and must not poison the rest. +pub(super) async fn r3(ch: &mut dyn Channel) -> Result<()> { + let (relay, target) = (0, 1); + let list = json!({ "Kind": "CollectionList" }); + ch.control(Verb::Stop(target)).await?; + let probe = ch.send(relay, target, Actor::Admin, list.clone()).await; + ch.control(Verb::Start(target)).await?; + let stopped = dial_check(probe, dial_budget_ms(ch.transport())); + ch.note(format!("stopped target: {}", verdict(&stopped))); + let ready_ms = await_ready(ch, target).await?; + let next = ch.send(relay, target, Actor::Admin, list).await; + ch.control(Verb::Regrant(target)).await?; + let (tail, restarted) = restart_check(next); + ch.note(format!( + "restarted target: ready after {ready_ms} ms; {tail}" + )); + stopped?; + restarted +} + +/// The restarted-target half, as its note tail and result: 200 via the +/// relay. A dial error once the target answers its own HTTP is the relay +/// failing to re-dial a peer it knew before the restart, not the grants a +/// 403 reports lost, and is named apart from that finding. +fn restart_check(next: Result) -> (String, Result<()>) { + let what = "admin CollectionList via the relay after the target restarted"; + if let Ok(r) = &next { + if let Some(e) = r.body["error"] + .as_str() + .filter(|e| e.contains("dial error")) + { + let got = format!( + "relay could not re-dial the restarted peer ({} {e}); \ + distinct from the restart NAC finding", + r.status + ); + return (got.clone(), Err(fail(format!("200 on {what}"), got))); + } + } + let checked = next.and_then(|r| { + expect_status(&r, 200, what) + .map(|()| "200 on admin CollectionList via the relay".to_string()) + }); + (verdict(&checked), checked.map(drop)) +} + +fn verdict(result: &Result) -> String { + match result { + Ok(text) => text.clone(), + Err(e) => format!("FAIL {e:#}"), + } +} + +/// Poll `node`'s own HTTP half a second apart until it answers or +/// [`READY_BUDGET`] passes; the milliseconds it took. +async fn await_ready(ch: &dyn Channel, node: usize) -> Result { + let start = tokio::time::Instant::now(); + loop { + let last = match ch.replicators(node) { + Ok(_) => return Ok(start.elapsed().as_millis() as u64), + Err(e) => e, + }; + if start.elapsed() >= READY_BUDGET { + return Err(last).wrap_err(format!( + "target not reachable {} ms after restart", + start.elapsed().as_millis() + )); + } + tokio::time::sleep(Duration::from_millis(500)).await; + } +} + +/// The stopped-target half: a clean 400 within `budget_ms`. +fn dial_check(probe: Result, budget_ms: u64) -> Result { + let probe = probe.map_err(|e| { + fail( + "a reply while the target is stopped", + format!("no reply: {e:#}"), + ) + })?; + expect_status(&probe, 400, "CollectionList to a stopped target")?; + if probe.latency_ms > budget_ms { + return Err(fail( + format!("a clean error within the {budget_ms} ms dial budget"), + format!("400 after {} ms: {}", probe.latency_ms, probe.body), + )); + } + Ok(format!( + "400 after {} ms, within the {budget_ms} ms dial budget", + probe.latency_ms + )) +} + +/// R2: the relay (node 0) has no replicator to the target (node 1); an admin +/// op still dials and lands. The relay's replicator is dropped and restored +/// through the channel in the other direction. +pub(super) async fn r2(ch: &mut dyn Channel) -> Result<()> { + let (relay, target) = (0, 1); + let target_addr = ch.addr(target); + let r = ch + .send(target, relay, Actor::Admin, replicator_delete(&target_addr)) + .await?; + expect_status(&r, 200, "ReplicatorDelete on the relay")?; + let list = ch + .send( + target, + relay, + Actor::Admin, + json!({ "Kind": "ReplicatorList" }), + ) + .await?; + let left = replicators_for(&list.body, &ch.peer_id(target)); + if left != 0 { + return Err(fail( + "no replicator from the relay to the target", + format!("{left} in {}", list.body), + )); + } + let r = ch + .send( + relay, + target, + Actor::Admin, + json!({ "Kind": "CollectionAdd", "collection_ids": [COLLECTION] }), + ) + .await?; + expect_status(&r, 200, "CollectionAdd via a relay without a replicator")?; + let list = ch + .send( + relay, + target, + Actor::Admin, + json!({ "Kind": "CollectionList" }), + ) + .await?; + if !list.body["values"] + .as_array() + .is_some_and(|v| v.iter().any(|c| c == COLLECTION)) + { + return Err(fail( + format!("{COLLECTION} in the target's CollectionList"), + list.body.to_string(), + )); + } + let r = ch + .send( + relay, + target, + Actor::Admin, + json!({ "Kind": "CollectionRemove", "collection_ids": [COLLECTION] }), + ) + .await?; + expect_status(&r, 200, "CollectionRemove (restore)")?; + let r = ch + .send(target, relay, Actor::Admin, replicator_add(&target_addr)) + .await?; + expect_status(&r, 200, "ReplicatorAdd on the relay (restore)") +} + +#[cfg(test)] +mod tests { + use std::collections::{BTreeSet, HashMap}; + + use super::super::cases::fake::*; + use super::super::cases::Outcome; + use super::*; + use serde_json::Value; + + /// A three-node mesh whose replicator lists follow the adds and deletes. + fn following_mesh() -> impl FnMut(usize, usize, usize, Actor, &Value) -> Result { + let mut mesh: HashMap> = (0..3) + .map(|t| { + ( + t, + (0..3) + .filter(|s| *s != t) + .map(|s| format!("peer{s}")) + .collect(), + ) + }) + .collect(); + move |_, target, _, _, op| { + let peer = op["addresses"][0] + .as_str() + .and_then(|a| a.rsplit("/p2p/").next()) + .unwrap_or_default() + .to_string(); + match op["Kind"].as_str().unwrap() { + "ReplicatorDelete" => { + mesh.get_mut(&target).unwrap().remove(&peer); + status(200) + } + "ReplicatorAdd" => { + mesh.get_mut(&target).unwrap().insert(peer); + status(200) + } + "ReplicatorList" => ok(json!({"Kind": "Replicators", "replicators": + mesh[&target].iter().map(|p| json!({"id": p, "collections": ["bafy-user"]})).collect::>() + })), + _ => admin_view(op), + } + } + } + + #[tokio::test] + async fn r1_drops_and_restores_every_target_source_pair_through_every_third_node() { + let (tx, rx) = std::sync::mpsc::channel(); + let mut mesh = following_mesh(); + let outcome = run_one("R1", move |relay, target, _, actor, op| { + if op["Kind"] == "ReplicatorAdd" { + let src = op["addresses"][0] + .as_str() + .unwrap() + .rsplit('/') + .next() + .unwrap() + .to_string(); + tx.send((relay, target, src)).unwrap(); + } + assert_eq!(actor, Actor::Admin); + mesh(relay, target, target, actor, op) + }) + .await; + assert_eq!(outcome, Outcome::Pass); + let mut triples: Vec<_> = rx.try_iter().collect(); + triples.sort(); + assert_eq!( + triples, + [ + (0, 1, "peer2".into()), + (0, 2, "peer1".into()), + (1, 0, "peer2".into()), + (1, 2, "peer0".into()), + (2, 0, "peer1".into()), + (2, 1, "peer0".into()), + ] + ); + } + + #[tokio::test] + async fn r1_fails_when_a_list_does_not_follow_the_op() { + let stuck = run_one("R1", |_, _, _, _, op| { + if op["Kind"] == "ReplicatorList" { + ok(json!({"Kind": "Replicators", "replicators": [ + {"id": "peer0"}, {"id": "peer1"}, {"id": "peer2"} + ]})) + } else { + status(200) + } + }) + .await; + assert!( + matches!(&stuck, Outcome::Fail { expected, got } if expected.contains("0 replicator") && expected.contains("ReplicatorDelete") && got.starts_with("1 in")), + "{stuck:?}" + ); + } + + /// While node 1 is stopped the relay answers `status` after `ms`. + fn stopped_target( + verbs: std::rc::Rc>>, + reply: Result, + ) -> Fake { + let reply = std::cell::Cell::new(Some(reply)); + let seen = verbs.clone(); + let mut fake = Fake::new(move |_, target, _, _, op| { + let down = seen.borrow().last() == Some(&Verb::Stop(1)); + if target == 1 && down { + return reply.take().expect("one probe while stopped"); + } + admin_view(op) + }); + fake.verbs = verbs; + fake + } + + fn after(status: u16, ms: u64) -> Result { + Ok(Reply { + status, + body: json!({"error": "dial timeout"}), + latency_ms: ms, + }) + } + + #[tokio::test] + async fn r3_wants_a_clean_400_within_the_dial_budget_then_a_restart() { + let verbs = std::rc::Rc::default(); + let (outcome, seen) = run_fake("R3", stopped_target(verbs, after(400, 9_800))).await; + assert_eq!(outcome, Outcome::Pass); + assert_eq!(seen, [Verb::Stop(1), Verb::Start(1), Verb::Regrant(1)]); + } + + #[tokio::test] + async fn r3_fails_on_a_slow_error_a_hang_or_a_200_and_still_restarts() { + let slow = run_fake("R3", stopped_target(Default::default(), after(400, 31_000))).await; + assert!( + matches!(&slow.0, Outcome::Fail { expected, got } if expected.contains("dial budget") && got.contains("31000 ms")), + "{:?}", + slow.0 + ); + assert_eq!(slow.1, [Verb::Stop(1), Verb::Start(1), Verb::Regrant(1)]); + + let hung = run_fake( + "R3", + stopped_target(Default::default(), Err(eyre::eyre!("operation timed out"))), + ) + .await; + assert!( + matches!(&hung.0, Outcome::Fail { got, .. } if got.contains("timed out")), + "{:?}", + hung.0 + ); + assert_eq!(hung.1, [Verb::Stop(1), Verb::Start(1), Verb::Regrant(1)]); + + let served = run_fake("R3", stopped_target(Default::default(), after(200, 5))).await; + assert!( + matches!(&served.0, Outcome::Fail { expected, .. } if expected.starts_with("400")), + "{:?}", + served.0 + ); + } + + #[tokio::test] + async fn r3_fails_when_the_restarted_target_refuses_admin_and_still_regrants() { + let verbs: std::rc::Rc>> = Default::default(); + let seen = verbs.clone(); + let mut fake = Fake::new(move |_, target, _, _, op| match seen.borrow().last() { + Some(Verb::Stop(1)) if target == 1 => after(400, 9_800), + Some(Verb::Start(1)) if target == 1 => status(403), + _ => admin_view(op), + }); + fake.verbs = verbs.clone(); + let (outcome, notes) = run_noted("R3", fake).await; + assert!( + matches!(&outcome, Outcome::Fail { expected, got } if expected.contains("restarted") && got.starts_with("403")), + "{outcome:?}" + ); + assert_eq!( + notes, + [ + "stopped target: 400 after 9800 ms, within the 15000 ms dial budget", + "restarted target: ready after 0 ms; FAIL expected 200 on admin CollectionList via the relay after the target restarted, got 403 null" + ] + ); + assert_eq!( + *verbs.borrow(), + [Verb::Stop(1), Verb::Start(1), Verb::Regrant(1)] + ); + } + + #[tokio::test] + async fn r3_names_a_relay_that_cannot_re_dial_the_restarted_target_apart_from_nac() { + let verbs: std::rc::Rc>> = Default::default(); + let seen = verbs.clone(); + let mut fake = Fake::new(move |_, target, _, _, op| match seen.borrow().last() { + Some(Verb::Stop(1)) if target == 1 => after(400, 9_800), + Some(Verb::Start(1)) if target == 1 => Ok(Reply { + status: 400, + body: json!({"error": "dial error: timed out"}), + latency_ms: 30_012, + }), + _ => admin_view(op), + }); + fake.verbs = verbs.clone(); + let (outcome, notes) = run_noted("R3", fake).await; + assert_eq!( + outcome, + Outcome::Fail { + expected: "200 on admin CollectionList via the relay after the target restarted" + .into(), + got: "relay could not re-dial the restarted peer (400 dial error: timed out); \ + distinct from the restart NAC finding" + .into(), + } + ); + assert_eq!( + notes, + [ + "stopped target: 400 after 9800 ms, within the 15000 ms dial budget", + "restarted target: ready after 0 ms; relay could not re-dial the restarted peer \ + (400 dial error: timed out); distinct from the restart NAC finding" + ] + ); + assert_eq!( + *verbs.borrow(), + [Verb::Stop(1), Verb::Start(1), Verb::Regrant(1)] + ); + } + + #[tokio::test] + async fn r3_notes_both_halves_and_gives_iroh_its_slow_dead_peer_dial() { + let (outcome, notes) = + run_noted("R3", stopped_target(Default::default(), after(400, 9_800))).await; + assert_eq!(outcome, Outcome::Pass); + assert_eq!( + notes, + [ + "stopped target: 400 after 9800 ms, within the 15000 ms dial budget", + "restarted target: ready after 0 ms; 200 on admin CollectionList via the relay" + ] + ); + + let mut iroh = stopped_target(Default::default(), after(400, 31_000)); + iroh.transport = Transport::Iroh; + let (outcome, notes) = run_noted("R3", iroh).await; + assert_eq!(outcome, Outcome::Pass, "{notes:?}"); + assert_eq!( + notes[0], + "stopped target: 400 after 31000 ms, within the 35000 ms dial budget" + ); + + let mut iroh = stopped_target(Default::default(), after(400, 36_000)); + iroh.transport = Transport::Iroh; + let (outcome, _) = run_noted("R3", iroh).await; + assert!( + matches!(&outcome, Outcome::Fail { expected, got } if expected.contains("35000 ms dial budget") && got.contains("36000 ms")), + "{outcome:?}" + ); + } + + /// A restarted node 1 that answers its own HTTP from poll `from` on. + fn ready_from(from: usize) -> impl FnMut(usize) -> Result { + let mut polls = 0; + move |node| { + if node != 1 { + return Ok(json!([])); + } + polls += 1; + if polls >= from { + Ok(json!([])) + } else { + eyre::bail!("connection refused") + } + } + } + + #[tokio::test(start_paused = true)] + async fn r3_waits_for_the_restarted_target_to_answer_its_own_http_then_checks() { + let verbs: std::rc::Rc>> = Default::default(); + let seen = verbs.clone(); + let ready = std::rc::Rc::new(std::cell::Cell::new(false)); + let up = ready.clone(); + let mut fake = Fake::new(move |_, target, _, _, op| match seen.borrow().last() { + Some(Verb::Stop(1)) if target == 1 => after(400, 9_800), + Some(Verb::Start(1)) if target == 1 && !up.get() => after(400, 30_012), + _ => admin_view(op), + }); + fake.verbs = verbs.clone(); + let mut polls = ready_from(3); + fake.own = std::cell::RefCell::new(Box::new(move |node| { + let r = polls(node); + ready.set(r.is_ok()); + r + })); + let (outcome, notes) = run_noted("R3", fake).await; + assert_eq!(outcome, Outcome::Pass, "{notes:?}"); + assert_eq!( + notes[1], + "restarted target: ready after 1000 ms; 200 on admin CollectionList via the relay" + ); + assert_eq!( + *verbs.borrow(), + [Verb::Stop(1), Verb::Start(1), Verb::Regrant(1)] + ); + } + + #[tokio::test(start_paused = true)] + async fn r3_is_infra_not_fail_when_the_restarted_target_never_answers() { + let mut fake = stopped_target(Default::default(), after(400, 9_800)); + fake.own = std::cell::RefCell::new(Box::new(|node| { + if node == 1 { + eyre::bail!("connection refused") + } + Ok(json!([])) + })); + let (outcome, notes) = run_noted("R3", fake).await; + assert!( + matches!(&outcome, Outcome::Infra { error } if error.starts_with("target not reachable 30000 ms after restart: ") && error.ends_with("connection refused")), + "{outcome:?}" + ); + assert_eq!( + notes, + ["stopped target: 400 after 9800 ms, within the 15000 ms dial budget"] + ); + } + + #[tokio::test] + async fn r2_drops_the_relay_replicator_before_the_probe_and_restores_it() { + let mut seen = Vec::new(); + let (tx, rx) = std::sync::mpsc::channel(); + let outcome = run_one("R2", move |relay, target, _, actor, op| { + tx.send(( + relay, + target, + actor, + op["Kind"].as_str().unwrap().to_string(), + )) + .unwrap(); + admin_view(op) + }) + .await; + assert_eq!(outcome, Outcome::Pass); + seen.extend(rx.try_iter().skip_while(|s| s.3 == "CollectionList")); + // The relay's own replicator is managed through the target as relay. + assert_eq!(seen[0], (1, 0, Actor::Admin, "ReplicatorDelete".into())); + assert_eq!(seen[1], (1, 0, Actor::Admin, "ReplicatorList".into())); + assert!(seen + .iter() + .any(|s| s == &(0, 1, Actor::Admin, "CollectionAdd".into()))); + assert_eq!( + seen.last().unwrap(), + &(1, 0, Actor::Admin, "ReplicatorAdd".into()) + ); + } + + #[tokio::test] + async fn r2_fails_when_the_relay_still_replicates_to_the_target() { + let outcome = run_one("R2", |_, target, _, _, op| { + if op["Kind"] == "ReplicatorList" && target == 0 { + ok(json!({"Kind": "Replicators", "replicators": [{"id": "peer1", "collections": ["bafy-user"]}]})) + } else { + admin_view(op) + } + }) + .await; + assert!( + matches!(&outcome, Outcome::Fail { expected, .. } if expected.contains("no replicator")), + "{outcome:?}" + ); + } +} diff --git a/crates/soak/src/manage/state.rs b/crates/soak/src/manage/state.rs new file mode 100644 index 0000000..3854c53 --- /dev/null +++ b/crates/soak/src/manage/state.rs @@ -0,0 +1,540 @@ +//! State cases: what a mutate leaves in the target's lists, and on the +//! data plane behind them. + +use eyre::Result; +use serde_json::{json, Value}; + +use super::actors::Actor; +use super::cases::{ + expect_status, fail, managed_state, replicator_add, replicator_delete, replicators_for, + Channel, Verb, COLLECTION, +}; +use super::data::{converge, ids_in, list_users, name_of, settle, update_user, write_docs}; + +/// A reply later than this is the correlator's 30 s, not the node's answer. +const CLEAN_MS: u64 = 5_000; + +/// A well-formed peer no node in the mesh has. +const ABSENT_PEER: &str = + "/ip4/127.0.0.1/tcp/1/p2p/12D3KooWQYhTNQdmr3ArTeUHRYzFg94BKyTkoWBDWez9kSCVe2Xo"; + +pub(super) fn replicator_add_filtered(addr: &str, filters: &Value) -> Value { + json!({ "Kind": "ReplicatorAdd", "addresses": [addr], "collection_ids": [COLLECTION], "filters": filters }) +} + +pub(super) fn document_add(id: &str) -> Value { + json!({ "Kind": "DocumentAdd", "docs": [{ "collection": COLLECTION, "doc_id": id }] }) +} + +pub(super) fn document_remove(id: &str) -> Value { + json!({ "Kind": "DocumentRemove", "docs": [{ "collection": COLLECTION, "doc_id": id }] }) +} + +/// The `filters` of the `Replicators` entry for `peer_id`. +fn filters_for(body: &Value, peer_id: &str) -> Value { + body["replicators"] + .as_array() + .into_iter() + .flatten() + .find(|r| r["id"] == peer_id) + .map_or(Value::Null, |r| r["filters"].clone()) +} + +/// Every replicator from `source`, to Rust and Go peers alike, dropped +/// (`add` false) or restored. A Go peer forwards what it receives to its +/// own replicators, so one left in place would leak the source's writes +/// around a filter. +async fn mesh_from(ch: &mut dyn Channel, relay: usize, source: usize, add: bool) -> Result<()> { + let peers = (0..ch.len()).chain(ch.go_nodes()); + for j in peers.filter(|j| *j != source) { + let addr = ch.addr(j); + let (op, name) = if add { + (replicator_add(&addr), "ReplicatorAdd") + } else { + (replicator_delete(&addr), "ReplicatorDelete") + }; + let r = ch.send(relay, source, Actor::Admin, op).await?; + expect_status(&r, 200, &format!("{name} on {source} for {j}"))?; + } + Ok(()) +} + +/// S2: `ReplicatorDelete` for a peer the target has no replicator to. The +/// node may say no (400) or nothing (200); either is fine when it comes +/// quickly and the lists are unchanged; which one is recorded. +pub(super) async fn s2(ch: &mut dyn Channel) -> Result<()> { + let (relay, target) = (0, 1); + let before = managed_state(ch, relay, target).await?; + let r = ch + .send(relay, target, Actor::Admin, replicator_delete(ABSENT_PEER)) + .await?; + let after = managed_state(ch, relay, target).await?; + if !matches!(r.status, 200 | 400) || r.latency_ms > CLEAN_MS { + return Err(fail( + format!("200 no-op or 400 error within {CLEAN_MS} ms for ReplicatorDelete of an absent peer"), + format!("{} after {} ms: {}", r.status, r.latency_ms, r.body), + )); + } + if before != after { + return Err(fail( + format!("target state unchanged: {before}"), + after.to_string(), + )); + } + ch.note(match r.status { + 200 => "ReplicatorDelete of an absent peer: 200 no-op".to_string(), + _ => format!("ReplicatorDelete of an absent peer: 400 {}", r.body), + }); + Ok(()) +} + +/// S3: the source's replicators are dropped and one to the sink comes back +/// through the channel with an `age` filter; documents of both ages are +/// written at the source and only the matching ones reach the sink. The +/// filter the list shows is then compared with one added on the source +/// itself. The mesh is restored either way. +pub(super) async fn s3(ch: &mut dyn Channel) -> Result<()> { + let (relay, source) = (0, 1); + let sink = relay; + let filters = json!({ COLLECTION: { "Field": "age", "Value": 1 } }); + let sink_addr = ch.addr(sink); + let sink_peer = ch.peer_id(sink); + let list = json!({ "Kind": "ReplicatorList" }); + mesh_from(ch, relay, source, false).await?; + let r = ch + .send( + relay, + source, + Actor::Admin, + replicator_add_filtered(&sink_addr, &filters), + ) + .await?; + expect_status(&r, 200, "ReplicatorAdd with a filter")?; + let remote = filters_for( + &ch.send(relay, source, Actor::Admin, list.clone()) + .await? + .body, + &sink_peer, + ); + let hit = write_docs(ch, source, 2, 1).await?; + let miss = write_docs(ch, source, 2, 2).await?; + let have = converge(ch, sink, &hit).await?; + let r = ch + .send(relay, source, Actor::Admin, replicator_delete(&sink_addr)) + .await?; + expect_status(&r, 200, "ReplicatorDelete of the filtered replicator")?; + ch.control(Verb::LocalReplicatorAdd { + node: source, + peer: sink, + filters: filters.clone(), + }) + .await?; + let local = filters_for( + &ch.send(relay, source, Actor::Admin, list).await?.body, + &sink_peer, + ); + let r = ch + .send(relay, source, Actor::Admin, replicator_delete(&sink_addr)) + .await?; + expect_status(&r, 200, "ReplicatorDelete of the local twin")?; + mesh_from(ch, relay, source, true).await?; + if !hit.iter().all(|id| have.contains(id)) { + return Err(fail( + format!("the sink to have the age-1 documents {hit:?}"), + format!("{have:?}"), + )); + } + if miss.iter().any(|id| have.contains(id)) { + return Err(fail( + format!("the sink without the age-2 documents {miss:?}"), + format!("{have:?}"), + )); + } + if remote != local { + return Err(fail( + format!("the relayed filter equal to the local one {local}"), + remote.to_string(), + )); + } + Ok(()) +} + +/// S4: with the source's replicators dropped, a document written at the +/// source stays there; `DocumentAdd` for it on the target (relayed by the +/// source, the only other node), then an update at the source, and the +/// target has it with the update. Restores the target's document list +/// and the mesh. +pub(super) async fn s4(ch: &mut dyn Channel) -> Result<()> { + let (relay, source) = (0, 1); + let target = relay; + mesh_from(ch, relay, source, false).await?; + let ids = write_docs(ch, source, 1, 1).await?; + let Some(id) = ids.first().cloned() else { + return Err(fail("one document id from the source", format!("{ids:?}"))); + }; + let r = ch + .send(source, target, Actor::Admin, document_add(&id)) + .await?; + expect_status(&r, 200, "DocumentAdd on the target")?; + ch.gql(source, update_user(&id, "touched")).await?; + let data = settle(ch, target, &list_users(), |data| { + name_of(data, &id).as_deref() == Some("touched") + }) + .await?; + let r = ch + .send(source, target, Actor::Admin, document_remove(&id)) + .await?; + expect_status(&r, 200, "DocumentRemove (restore)")?; + mesh_from(ch, relay, source, true).await?; + let expected = format!( + "the target to have {id} named touched after DocumentAdd and an update at the source" + ); + match name_of(&data, &id).as_deref() { + Some("touched") => Ok(()), + Some(name) => Err(fail( + expected, + format!("{id} named {name}: the document without the update"), + )), + None => Err(fail( + expected, + format!( + "no document {id} among {} on the target", + ids_in(&data, COLLECTION).len() + ), + )), + } +} + +/// S1: `ReplicatorAdd` twice for the same peer leaves one entry. The mesh +/// replicator from the target to the relay is dropped first so the first +/// add is a real add; the second add restores the mesh. +pub(super) async fn s1(ch: &mut dyn Channel) -> Result<()> { + let (relay, target) = (0, 1); + let relay_addr = ch.addr(relay); + let r = ch + .send(relay, target, Actor::Admin, replicator_delete(&relay_addr)) + .await?; + expect_status(&r, 200, "ReplicatorDelete before the double add")?; + for n in 1..=2 { + let r = ch + .send(relay, target, Actor::Admin, replicator_add(&relay_addr)) + .await?; + expect_status(&r, 200, &format!("ReplicatorAdd #{n}"))?; + } + let list = ch + .send( + relay, + target, + Actor::Admin, + json!({ "Kind": "ReplicatorList" }), + ) + .await?; + expect_status(&list, 200, "ReplicatorList")?; + let entries = replicators_for(&list.body, &ch.peer_id(relay)); + if entries != 1 { + return Err(fail( + "one replicator entry for the relay after two adds", + format!("{entries} in {}", list.body), + )); + } + Ok(()) +} + +#[cfg(test)] +mod tests { + use std::cell::RefCell; + + use super::super::cases::fake::*; + use super::super::cases::Outcome; + use super::super::client::Reply; + use super::super::data::fake::store; + use super::*; + + #[tokio::test] + async fn s1_adds_twice_and_wants_one_entry() { + let mut adds = 0; + let (tx, rx) = std::sync::mpsc::channel(); + let outcome = run_one("S1", move |_, _, _, _, op| { + if op["Kind"] == "ReplicatorAdd" { + adds += 1; + tx.send(adds).unwrap(); + } + admin_view(op) + }) + .await; + assert_eq!(outcome, Outcome::Pass); + assert_eq!(rx.try_iter().last(), Some(2)); + + let doubled = run_one("S1", |_, _, _, _, op| { + if op["Kind"] == "ReplicatorList" { + ok(json!({"Kind": "Replicators", "replicators": [{"id": "peer0"}, {"id": "peer0"}]})) + } else { + admin_view(op) + } + }) + .await; + assert!( + matches!(&doubled, Outcome::Fail { got, .. } if got.contains('2')), + "{doubled:?}" + ); + } + + fn replied(status: u16, ms: u64, body: Value) -> Result { + Ok(Reply { + status, + body, + latency_ms: ms, + }) + } + + #[tokio::test] + async fn s2_takes_a_quick_no_op_or_error_and_records_it() { + for (status, want) in [(200, "200 no-op"), (400, "400 {\"error\":\"not found\"}")] { + let fake = Fake::new(move |_, _, _, _, op| { + if op["Kind"] == "ReplicatorDelete" { + assert_eq!(op["addresses"][0], ABSENT_PEER); + replied(status, 12, json!({"error": "not found"})) + } else { + admin_view(op) + } + }); + let (outcome, notes) = run_noted("S2", fake).await; + assert_eq!(outcome, Outcome::Pass, "{status}"); + assert_eq!( + notes, + [format!("ReplicatorDelete of an absent peer: {want}")] + ); + } + } + + #[tokio::test] + async fn s2_fails_on_a_5xx_a_late_reply_or_a_changed_list() { + let server_error = run_one("S2", |_, _, _, _, op| match op["Kind"].as_str() { + Some("ReplicatorDelete") => replied(500, 12, Value::Null), + _ => admin_view(op), + }) + .await; + assert!( + matches!(&server_error, Outcome::Fail { got, .. } if got.starts_with("500")), + "{server_error:?}" + ); + let late = run_one("S2", |_, _, _, _, op| match op["Kind"].as_str() { + Some("ReplicatorDelete") => replied(400, 30_100, json!({"error": "response timeout"})), + _ => admin_view(op), + }) + .await; + assert!( + matches!(&late, Outcome::Fail { got, .. } if got.contains("30100 ms")), + "{late:?}" + ); + let lists = RefCell::new(0); + let changed = run_one("S2", move |_, _, _, _, op| match op["Kind"].as_str() { + Some("ReplicatorDelete") => status(200), + Some("ReplicatorList") => { + *lists.borrow_mut() += 1; + if *lists.borrow() > 1 { + ok(json!({"Kind": "Replicators", "replicators": []})) + } else { + admin_view(op) + } + } + _ => admin_view(op), + }) + .await; + assert!( + matches!(&changed, Outcome::Fail { expected, .. } if expected.contains("unchanged")), + "{changed:?}" + ); + } + + /// The source (node 1) keeps its replicator list, with filters, as + /// the ops shape it; `local_filters` is what the local add stores. + fn source_lists(local_filters: Value) -> Fake { + let entries: std::rc::Rc>> = Default::default(); + let seen = entries.clone(); + let verbs: std::rc::Rc>> = Default::default(); + let by_verb = verbs.clone(); + let applied = std::cell::Cell::new(0); + let mut fake = Fake::new(move |_, target, _, _, op| { + let peer = op["addresses"][0] + .as_str() + .and_then(|a| a.rsplit("/p2p/").next()) + .unwrap_or_default() + .to_string(); + for verb in by_verb.borrow().iter().skip(applied.get()) { + if let Verb::LocalReplicatorAdd { peer, .. } = verb { + seen.borrow_mut() + .push(json!({"id": format!("peer{peer}"), "filters": local_filters})); + } + } + applied.set(by_verb.borrow().len()); + match op["Kind"].as_str().unwrap() { + "ReplicatorDelete" if target == 1 => { + seen.borrow_mut().retain(|r| r["id"] != peer); + status(200) + } + "ReplicatorAdd" if target == 1 => { + seen.borrow_mut() + .push(json!({"id": peer, "filters": op["filters"]})); + status(200) + } + "ReplicatorList" if target == 1 => { + ok(json!({"Kind": "Replicators", "replicators": *seen.borrow()})) + } + _ => admin_view(op), + } + }); + fake.verbs = verbs; + fake + } + + #[tokio::test] + async fn s3_wants_only_matching_docs_at_the_sink_and_the_same_filter_both_ways() { + let filters = json!({"User": {"Field": "age", "Value": 1}}); + let mut fake = source_lists(filters.clone()); + fake.gql = RefCell::new(Box::new(store(|age| age == 1))); + let (outcome, verbs) = run_fake("S3", fake).await; + assert_eq!(outcome, Outcome::Pass); + assert_eq!( + verbs, + [Verb::LocalReplicatorAdd { + node: 1, + peer: 0, + filters + }] + ); + + let mut leaky = source_lists(json!({"User": {"Field": "age", "Value": 1}})); + leaky.gql = RefCell::new(Box::new(store(|_| true))); + let (outcome, _) = run_fake("S3", leaky).await; + assert!( + matches!(&outcome, Outcome::Fail { expected, .. } if expected.contains("without the age-2")), + "{outcome:?}" + ); + + let mut mangled = source_lists(json!({"User": {"Field": "age", "Value": "1"}})); + mangled.gql = RefCell::new(Box::new(store(|age| age == 1))); + let (outcome, _) = run_fake("S3", mangled).await; + assert!( + matches!(&outcome, Outcome::Fail { expected, .. } if expected.contains("equal to the local")), + "{outcome:?}" + ); + } + + #[tokio::test(start_paused = true)] + async fn s3_fails_when_the_sink_never_converges_and_still_restores_the_mesh() { + let (tx, rx) = std::sync::mpsc::channel(); + let mut fake = source_lists(json!({"User": {"Field": "age", "Value": 1}})); + let inner = fake.rule.replace(Box::new(|_, _, _, _, _| status(200))); + let inner = RefCell::new(inner); + fake.rule = RefCell::new(Box::new(move |r, t, a, actor, op| { + tx.send((t, op["Kind"].as_str().unwrap().to_string())) + .unwrap(); + (inner.borrow_mut())(r, t, a, actor, op) + })); + fake.gql = RefCell::new(Box::new(store(|_| false))); + let (outcome, _) = run_fake("S3", fake).await; + assert!( + matches!(&outcome, Outcome::Fail { expected, .. } if expected.contains("to have the age-1")), + "{outcome:?}" + ); + let ops: Vec<_> = rx.try_iter().collect(); + let adds = ops + .iter() + .filter(|(t, k)| *t == 1 && k == "ReplicatorAdd") + .count(); + assert_eq!(adds, 3, "one filtered add, two restoring the mesh: {ops:?}"); + } + + #[tokio::test] + async fn s3_drops_and_restores_the_source_replicators_to_the_go_nodes_too() { + let (tx, rx) = std::sync::mpsc::channel(); + let mut fake = source_lists(json!({"User": {"Field": "age", "Value": 1}})); + fake.go = vec![3]; + let inner = fake.rule.replace(Box::new(|_, _, _, _, _| status(200))); + let inner = RefCell::new(inner); + fake.rule = RefCell::new(Box::new(move |r, t, a, actor, op| { + tx.send(( + t, + op["Kind"].as_str().unwrap().to_string(), + op["addresses"][0].clone(), + )) + .unwrap(); + (inner.borrow_mut())(r, t, a, actor, op) + })); + fake.gql = RefCell::new(Box::new(store(|age| age == 1))); + let (outcome, _) = run_fake("S3", fake).await; + assert_eq!(outcome, Outcome::Pass); + let ops: Vec<_> = rx.try_iter().collect(); + let to_go = |kind: &str| { + ops.iter() + .filter(|(t, k, a)| *t == 1 && k == kind && a == &fake_addr(3)) + .count() + }; + assert_eq!( + (to_go("ReplicatorDelete"), to_go("ReplicatorAdd")), + (1, 1), + "{ops:?}" + ); + } + + fn fake_addr(node: usize) -> Value { + json!(format!("/ip4/127.0.0.1/tcp/{node}/p2p/peer{node}")) + } + + #[tokio::test] + async fn s4_adds_the_doc_on_the_target_via_the_source_then_updates_and_wants_it_there() { + let (tx, rx) = std::sync::mpsc::channel(); + let mut fake = Fake::new(move |relay, target, _, _, op| { + tx.send(( + relay, + target, + op["Kind"].as_str().unwrap().to_string(), + op["docs"][0]["doc_id"].clone(), + )) + .unwrap(); + admin_view(op) + }); + fake.gql = RefCell::new(Box::new(store(|_| true))); + let (outcome, _) = run_fake("S4", fake).await; + assert_eq!(outcome, Outcome::Pass); + let ops: Vec<_> = rx.try_iter().collect(); + assert!( + ops.contains(&(1, 0, "DocumentAdd".into(), json!("bae-0"))), + "{ops:?}" + ); + assert!( + ops.contains(&(1, 0, "DocumentRemove".into(), json!("bae-0"))), + "{ops:?}" + ); + assert_eq!(ops.last().unwrap().2, "ReplicatorAdd", "{ops:?}"); + } + + #[tokio::test(start_paused = true)] + async fn s4_fails_when_the_target_never_gets_the_doc() { + let mut fake = Fake::new(|_, _, _, _, op| admin_view(op)); + fake.gql = RefCell::new(Box::new(store(|_| false))); + let (outcome, _) = run_fake("S4", fake).await; + assert!( + matches!(&outcome, Outcome::Fail { expected, got } if expected.contains("bae-0 named touched") && got == "no document bae-0 among 0 on the target"), + "{outcome:?}" + ); + } + + #[tokio::test(start_paused = true)] + async fn s4_fails_when_the_target_has_the_doc_without_the_update() { + let mut fake = Fake::new(|_, _, _, _, op| admin_view(op)); + let mut inner = store(|_| true); + fake.gql = RefCell::new(Box::new(move |node, q| { + if q.starts_with("mutation { update_") { + return Ok(json!({})); + } + inner(node, q) + })); + let (outcome, _) = run_fake("S4", fake).await; + assert!( + matches!(&outcome, Outcome::Fail { expected, got } if expected.contains("bae-0 named touched") && got.starts_with("bae-0 named u") && got.ends_with(": the document without the update")), + "{outcome:?}" + ); + } +} diff --git a/crates/soak/src/meter.rs b/crates/soak/src/meter.rs new file mode 100644 index 0000000..c174bd9 --- /dev/null +++ b/crates/soak/src/meter.rs @@ -0,0 +1,261 @@ +//! Disk and memory meter plus the write-budget governor. +//! +//! Every write grows the Merkle DAG on every node forever, so a soak has a +//! disk ceiling. Each sample writes `du.jsonl` and `rss.jsonl`, then the +//! governor turns the remaining budget into the op rate that would exhaust +//! it exactly at the deadline, clamped to `[floor, profile rate]`, and +//! raises the hard stop at 95% of the ceiling. Amplification is measured +//! live as mesh-wide bytes grown per executed op since the first sample. + +use std::fs::File; +use std::io::{BufWriter, Write}; +use std::path::Path; +use std::process::Command; +use std::sync::atomic::{AtomicBool, AtomicU64, Ordering}; +use std::sync::Arc; +use std::time::{Duration, Instant}; + +use eyre::{Result, WrapErr}; +use serde_json::json; + +use crate::executor::now_ms; +use crate::nodes::Nodes; + +/// Rate that spends `remaining_bytes` over `remaining_secs` at +/// `bytes_per_op` (mesh-wide growth per executed op), clamped to +/// `[floor, ceiling_rate]`. A non-positive or unknown `bytes_per_op` means +/// nothing measured yet: full rate. +pub fn governed_rate( + remaining_bytes: f64, + bytes_per_op: f64, + remaining_secs: f64, + floor: f64, + ceiling_rate: f64, +) -> f64 { + if bytes_per_op.is_nan() || bytes_per_op <= 0.0 || remaining_secs <= 0.0 { + return ceiling_rate; + } + let allowed = remaining_bytes.max(0.0) / (bytes_per_op * remaining_secs); + allowed.clamp(floor.min(ceiling_rate), ceiling_rate) +} + +/// Nearest-rank percentile of an unsorted sample; `None` when empty. +pub fn percentile(values: &[u64], p: f64) -> Option { + if values.is_empty() { + return None; + } + let mut sorted = values.to_vec(); + sorted.sort_unstable(); + let rank = (p * sorted.len() as f64).ceil() as usize; + Some(sorted[rank.clamp(1, sorted.len()) - 1]) +} + +pub struct MeterConfig { + pub interval: Duration, + pub ceiling_bytes: u64, + pub floor_rate: f64, + pub profile_rate: f64, + /// Wall deadline of a `--secs` run; op-count runs derive one from the + /// remaining ops at the profile rate. + pub deadline: Option, + pub ops: usize, +} + +pub struct Meter { + cfg: MeterConfig, + du: BufWriter, + rss: BufWriter, + budget: BufWriter, + /// (total bytes, op index) at the first sample. + baseline: Option<(u64, u64)>, + last_sample: Option, + last_rate_milli: u64, + /// Governed op rate x1000, read by the workload loop. + rate_milli: Arc, + /// Hard stop, read by the workload loop. + stop: Arc, + op_index: Arc, +} + +impl Meter { + pub fn new( + cfg: MeterConfig, + run_dir: &Path, + rate_milli: Arc, + stop: Arc, + op_index: Arc, + ) -> Result { + let open = |name: &str| -> Result> { + let path = run_dir.join(name); + Ok(BufWriter::new(File::create(&path).wrap_err_with(|| { + format!("creating {}", path.display()) + })?)) + }; + Ok(Self { + du: open("du.jsonl")?, + rss: open("rss.jsonl")?, + budget: open("budget.jsonl")?, + baseline: None, + last_sample: None, + last_rate_milli: (cfg.profile_rate * 1000.0) as u64, + cfg, + rate_milli, + stop, + op_index, + }) + } + + /// Sample if the interval elapsed (always on the first call). + pub async fn maybe_sample(&mut self, nodes: &Nodes) -> Result<()> { + if self + .last_sample + .is_some_and(|t| t.elapsed() < self.cfg.interval) + { + return Ok(()); + } + self.sample(nodes).await + } + + // ponytail: `du` and `ps` run synchronously on the task the workload + // shares, stalling op dispatch for the sample's duration (ms now, + // seconds near a 120 GiB ceiling); move to spawn_blocking when it shows. + pub async fn sample(&mut self, nodes: &Nodes) -> Result<()> { + self.last_sample = Some(Instant::now()); + let wall = now_ms(); + let op_index = self.op_index.load(Ordering::Relaxed); + let mut total = 0u64; + for (i, rss) in nodes.rss_all().await.into_iter().enumerate() { + let name = nodes.name(i); + let bytes = du_bytes(&nodes.rootdir(i))?; + total += bytes; + let line = + json!({"wall_ts_ms": wall, "op_index": op_index, "node": name, "bytes": bytes}); + serde_json::to_writer(&mut self.du, &line)?; + self.du.write_all(b"\n")?; + let pid = nodes.pid(i); + if pid.is_some() || rss.is_some() { + let line = json!({ + "wall_ts_ms": wall, "op_index": op_index, "node": name, "pid": pid, + "rss_bytes": rss, + }); + serde_json::to_writer(&mut self.rss, &line)?; + self.rss.write_all(b"\n")?; + } + } + self.du.flush()?; + self.rss.flush()?; + + let (base_bytes, base_op) = *self.baseline.get_or_insert((total, op_index)); + let bytes_per_op = if op_index > base_op { + total.saturating_sub(base_bytes) as f64 / (op_index - base_op) as f64 + } else { + 0.0 + }; + let remaining_secs = match self.cfg.deadline { + Some(d) => d.saturating_duration_since(Instant::now()).as_secs_f64(), + None => self.cfg.ops.saturating_sub(op_index as usize) as f64 / self.cfg.profile_rate, + }; + let remaining = self.cfg.ceiling_bytes as f64 - total as f64; + let rate = governed_rate( + remaining, + bytes_per_op, + remaining_secs, + self.cfg.floor_rate, + self.cfg.profile_rate, + ); + let rate_milli = (rate * 1000.0) as u64; + self.rate_milli.store(rate_milli, Ordering::Relaxed); + let hard_stop = total as f64 >= 0.95 * self.cfg.ceiling_bytes as f64; + if hard_stop && !self.stop.swap(true, Ordering::Relaxed) { + println!( + "budget: hard stop, {} of {} bytes used (95% ceiling)", + total, self.cfg.ceiling_bytes + ); + } + if rate_milli != self.last_rate_milli { + println!( + "budget: {:.1}% of ceiling used, {:.0} bytes/op, rate {:.2} -> {:.2} ops/s", + 100.0 * total as f64 / self.cfg.ceiling_bytes as f64, + bytes_per_op, + self.last_rate_milli as f64 / 1000.0, + rate + ); + self.last_rate_milli = rate_milli; + } + let line = json!({ + "wall_ts_ms": wall, "op_index": op_index, "total_bytes": total, + "ceiling_bytes": self.cfg.ceiling_bytes, "bytes_per_op": bytes_per_op, + "remaining_secs": remaining_secs, "rate": rate, "hard_stop": hard_stop, + }); + serde_json::to_writer(&mut self.budget, &line)?; + self.budget.write_all(b"\n")?; + self.budget.flush()?; + Ok(()) + } +} + +fn du_bytes(path: &Path) -> Result { + let out = Command::new("du") + .args(["-sk"]) + .arg(path) + .output() + .wrap_err("running du")?; + let text = String::from_utf8_lossy(&out.stdout); + let kb: u64 = text + .split_whitespace() + .next() + .and_then(|s| s.parse().ok()) + .unwrap_or(0); + Ok(kb * 1024) +} + +pub fn rss_bytes(pid: u32) -> Option { + let out = Command::new("ps") + .args(["-o", "rss=", "-p", &pid.to_string()]) + .output() + .ok()?; + String::from_utf8_lossy(&out.stdout) + .trim() + .parse::() + .ok() + .map(|kb| kb * 1024) +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn plenty_of_budget_runs_at_profile_rate() { + // 100 MiB left, 3 KiB per op, one hour: could do ~9 ops/s. + assert_eq!(governed_rate(100e6, 3000.0, 3600.0, 0.5, 5.0), 5.0); + } + + #[test] + fn tight_budget_scales_the_rate_down() { + // 10 MB left, 5 KB per op, 1000s: 2 ops/s spends it exactly. + let r = governed_rate(10e6, 5000.0, 1000.0, 0.5, 5.0); + assert!((r - 2.0).abs() < 1e-9, "{r}"); + } + + #[test] + fn never_below_the_floor() { + assert_eq!(governed_rate(1.0, 5000.0, 1000.0, 0.5, 5.0), 0.5); + assert_eq!(governed_rate(0.0, 5000.0, 1000.0, 0.5, 5.0), 0.5); + } + + #[test] + fn unmeasured_amplification_means_full_rate() { + assert_eq!(governed_rate(10e6, 0.0, 1000.0, 0.5, 5.0), 5.0); + assert_eq!(governed_rate(10e6, f64::NAN, 1000.0, 0.5, 5.0), 5.0); + } + + #[test] + fn percentiles_nearest_rank() { + let v = [50, 10, 40, 20, 30]; + assert_eq!(percentile(&v, 0.5), Some(30)); + assert_eq!(percentile(&v, 0.95), Some(50)); + assert_eq!(percentile(&v, 0.0), Some(10)); + assert_eq!(percentile(&[], 0.5), None); + } +} diff --git a/crates/soak/src/nodes.rs b/crates/soak/src/nodes.rs new file mode 100644 index 0000000..21c872f --- /dev/null +++ b/crates/soak/src/nodes.rs @@ -0,0 +1,912 @@ +//! Node-control seam: the same soak driver runs against harness-spawned +//! processes or docker containers on a user network. + +use std::collections::HashMap; +use std::path::{Path, PathBuf}; +use std::time::{Duration, Instant, SystemTime, UNIX_EPOCH}; + +use defra_harness::{extract_p2p_addr, DefraClient, StoppedNode, TestCluster}; +use eyre::{bail, ensure, Result, WrapErr}; +use serde::{Deserialize, Serialize}; + +/// The harness's file-keyring secret (`defra_harness::cluster::builder`). +const KEYRING_SECRET: &str = "integration-test-secret"; +const API_PORT: &str = "9181"; +const P2P_PORT: &str = "9171"; +const STOP_TIMEOUT: &str = "10"; +const READY_TIMEOUT: Duration = Duration::from_secs(90); + +#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)] +#[serde(rename_all = "snake_case")] +pub enum NodeKind { + Rust, + Go, +} + +impl NodeKind { + fn harness(self) -> defra_harness::NodeKind { + match self { + NodeKind::Rust => defra_harness::NodeKind::Rust, + NodeKind::Go => defra_harness::NodeKind::Go, + } + } + + /// Host-side CLI used to talk to a node of this kind. + pub fn host_binary(self) -> Result { + Ok(match self { + NodeKind::Rust => PathBuf::from( + std::env::var("DEFRA_RUST_BINARY").wrap_err("DEFRA_RUST_BINARY must be set")?, + ), + NodeKind::Go => PathBuf::from("defradb"), + }) + } +} + +#[derive(Debug, Clone)] +pub struct NodeSpec { + pub name: String, + pub kind: NodeKind, + pub store: String, + /// Partition side, "A" or "B". + pub host: &'static str, +} + +fn spec(name: &str, kind: NodeKind, store: &str, host: &'static str) -> NodeSpec { + NodeSpec { + name: name.into(), + kind, + store: store.into(), + host, + } +} + +/// The M2 topology: three nodes per partition side, both runtimes on each. +pub fn m2_specs() -> Vec { + vec![ + spec("rust-0", NodeKind::Rust, "regolith", "A"), + spec("rust-1", NodeKind::Rust, "regolith", "A"), + spec("go-0", NodeKind::Go, "badger", "A"), + spec("rust-2", NodeKind::Rust, "regolith", "B"), + spec("go-1", NodeKind::Go, "badger", "B"), + spec("go-2", NodeKind::Go, "badger", "B"), + ] +} + +/// `rust_n` Rust nodes then `go_n` Go nodes, each runtime's nodes alternating +/// between the two partition sides so neither side is single-runtime. Either +/// count may be zero: a homogeneous mesh is the control for a mixed one. +pub fn topology_specs(rust_n: usize, go_n: usize) -> Vec { + let side = |i: usize| if i.is_multiple_of(2) { "A" } else { "B" }; + (0..rust_n) + .map(|i| spec(&format!("rust-{i}"), NodeKind::Rust, "regolith", side(i))) + .chain((0..go_n).map(|i| spec(&format!("go-{i}"), NodeKind::Go, "badger", side(i)))) + .collect() +} + +#[derive(Debug, Clone)] +pub struct Container { + pub spec: NodeSpec, + pub image: String, + /// Host port published to the container's API port. + pub api_port: u16, + /// Host directory mounted at `/data`. + pub rootdir: PathBuf, + pub ip: Option, + pub peer_addr: Option, +} + +impl Container { + fn api_url(&self) -> String { + format!("http://127.0.0.1:{}", self.api_port) + } +} + +/// `docker run` arguments for `c`; the flag order per runtime mirrors the +/// harness builders (`rust_node.rs` / `go_node.rs`). `retry_intervals` is the +/// comma-separated `--replicator-retry-intervals` both runtimes take on +/// `start`; without it a container churn measures the default ladder, not +/// replication. +/// `node_env` is passed through as `-e KEY=VALUE` on every container. +pub fn docker_run_argv( + c: &Container, + network: &str, + secret: &str, + retry_intervals: Option<&str>, + node_env: &[String], +) -> Vec { + let mut v: Vec = [ + "run", + "-d", + "--name", + &format!("{network}-{}", c.spec.name), + "--network", + network, + "-p", + &format!("127.0.0.1:{}:{API_PORT}", c.api_port), + "-v", + &format!("{}:/data", c.rootdir.display()), + "-e", + &format!("DEFRA_KEYRING_SECRET={secret}"), + ] + .map(String::from) + .to_vec(); + for kv in node_env { + v.push("-e".into()); + v.push(kv.clone()); + } + v.extend([c.image.clone(), "--rootdir".into(), "/data".into()]); + let url = ["--url", &format!("0.0.0.0:{API_PORT}")].map(String::from); + let keyring = ["--keyring-backend", "file", "--keyring-path", "/data/keys"].map(String::from); + match c.spec.kind { + NodeKind::Rust => { + v.extend(url); + v.push("--no-log-color".into()); + v.extend(keyring); + v.push("start".into()); + } + NodeKind::Go => { + v.push("--no-log-color".into()); + v.extend(keyring); + v.push("start".into()); + v.extend(url); + } + } + v.extend( + [ + "--store", + &c.spec.store, + "--no-telemetry", + "--no-encryption", + "--no-searchable-encryption", + "--no-signing", + "--p2paddr", + &format!("/ip4/0.0.0.0/tcp/{P2P_PORT}"), + ] + .map(String::from), + ); + if let Some(intervals) = retry_intervals { + v.push("--replicator-retry-intervals".into()); + v.push(intervals.into()); + } + v +} + +/// Bytes from the usage half of a `docker stats` MEM USAGE / LIMIT cell. +pub fn parse_mem_usage(s: &str) -> Option { + let used = s.split('/').next()?.trim(); + let digits = used + .find(|ch: char| !(ch.is_ascii_digit() || ch == '.')) + .unwrap_or(used.len()); + let value: f64 = used[..digits].parse().ok()?; + let unit: f64 = match used[digits..].trim() { + "B" => 1.0, + "kB" => 1e3, + "MB" => 1e6, + "GB" => 1e9, + "KiB" => 1024.0, + "MiB" => 1024.0 * 1024.0, + "GiB" => 1024.0 * 1024.0 * 1024.0, + _ => return None, + }; + Some((value * unit) as u64) +} + +/// Run one `docker` command and return its stdout. `DOCKER_CONTEXT` is +/// inherited from the environment. +async fn docker(args: &[&str]) -> Result { + let out = tokio::process::Command::new("docker") + .args(args) + .output() + .await + .wrap_err_with(|| format!("docker {args:?}"))?; + let stderr = String::from_utf8_lossy(&out.stderr); + ensure!(out.status.success(), "docker {args:?}: {}", stderr.trim()); + Ok(String::from_utf8_lossy(&out.stdout).into_owned()) +} + +/// Names of every `soak-*` network on the docker host. +/// `kill `; `-STOP` freezes a node with its connections up, +/// `-CONT` lets it go on. +pub fn signal(pid: u32, sig: &str) -> Result<()> { + let out = std::process::Command::new("kill") + .args([sig, &pid.to_string()]) + .output() + .wrap_err_with(|| format!("kill {sig} {pid}"))?; + ensure!( + out.status.success(), + "kill {sig} {pid}: {}", + String::from_utf8_lossy(&out.stderr).trim() + ); + Ok(()) +} + +pub async fn existing_networks() -> Result> { + let out = docker(&[ + "network", + "ls", + "--filter", + "name=soak-", + "--format", + "{{.Name}}", + ]) + .await?; + Ok(out.lines().map(String::from).collect()) +} + +fn free_port() -> Result { + Ok(std::net::TcpListener::bind("127.0.0.1:0")? + .local_addr()? + .port()) +} + +fn unix_secs() -> u64 { + SystemTime::now() + .duration_since(UNIX_EPOCH) + .map(|d| d.as_secs()) + .unwrap_or(0) +} + +pub struct DockerNodes { + network: String, + containers: Vec, + http: reqwest::Client, + secret: String, + /// Unix seconds of the last `dump_logs` per container. + last_dump: Vec, +} + +impl DockerNodes { + /// Create the `soak-` network, then run and wait for every spec. + pub async fn start( + run_id: &str, + run_dir: &Path, + specs: Vec, + images: (&str, &str), + retry_intervals: Option<&str>, + node_env: &[String], + ) -> Result { + let network = format!("soak-{run_id}"); + docker(&["network", "create", &network]).await?; + let mut nodes = Self { + network, + containers: Vec::with_capacity(specs.len()), + http: reqwest::Client::new(), + secret: KEYRING_SECRET.to_string(), + last_dump: vec![0; specs.len()], + }; + let started = async { + for spec in specs { + let rootdir = run_dir.join("target").join("docker").join(&spec.name); + std::fs::create_dir_all(&rootdir) + .wrap_err_with(|| format!("creating {}", rootdir.display()))?; + let image = match spec.kind { + NodeKind::Rust => images.0, + NodeKind::Go => images.1, + }; + nodes.containers.push(Container { + spec, + image: image.to_string(), + api_port: free_port()?, + rootdir, + ip: None, + peer_addr: None, + }); + let i = nodes.containers.len() - 1; + let argv = docker_run_argv( + &nodes.containers[i], + &nodes.network, + &nodes.secret, + retry_intervals, + node_env, + ); + let argv: Vec<&str> = argv.iter().map(String::as_str).collect(); + docker(&argv).await?; + nodes.wait_ready(i).await?; + } + Ok(()) + } + .await; + match started { + Ok(()) => Ok(nodes), + Err(e) => { + nodes.teardown().await.ok(); + Err(e) + } + } + } + + /// Best-effort: dump every container's remaining logs, remove every + /// container, then the network; the first error is returned at the end. + /// A container that was never created (a partial `start`) is not an error. + async fn teardown(&mut self) -> Result<()> { + let mut first_err = None; + for i in 0..self.containers.len() { + if let Err(e) = self.dump_logs(i).await { + first_err.get_or_insert(e); + } + if let Err(e) = docker(&["rm", "-f", &self.container_name(i)]).await { + if !e.to_string().contains("No such container") { + first_err.get_or_insert(e); + } + } + } + if let Err(e) = docker(&["network", "rm", &self.network]).await { + first_err.get_or_insert(e); + } + first_err.map_or(Ok(()), Err) + } + + fn container_name(&self, i: usize) -> String { + format!("{}-{}", self.network, self.containers[i].spec.name) + } + + /// Poll `p2p/info` until it answers (both runtimes do once up; Go + /// rejects `{ __typename }`), then record the container's addresses. + async fn wait_ready(&mut self, i: usize) -> Result<()> { + let url = self.containers[i].api_url(); + let deadline = Instant::now() + READY_TIMEOUT; + while crate::churn::peer_id(&self.http, &url).await.is_none() { + ensure!( + Instant::now() < deadline, + "{}: API not ready within {READY_TIMEOUT:?}", + self.container_name(i) + ); + tokio::time::sleep(Duration::from_millis(500)).await; + } + self.record_addr(i).await + } + + /// Record the container's address on the soak network and its libp2p + /// peer address. The node reports `/ip4/0.0.0.0/...`, so the host part + /// is the inspected address; `rejoin` pins it, so it must not change. + async fn record_addr(&mut self, i: usize) -> Result<()> { + let name = self.container_name(i); + let url = self.containers[i].api_url(); + let ip = docker(&[ + "inspect", + "-f", + &format!( + "{{{{(index .NetworkSettings.Networks \"{}\").IPAddress}}}}", + self.network + ), + &name, + ]) + .await? + .trim() + .to_string(); + ensure!(!ip.is_empty(), "{name}: no address on {}", self.network); + if let Some(recorded) = &self.containers[i].ip { + ensure!( + *recorded == ip, + "{name}: rejoined at {ip}, peers hold {recorded}" + ); + } + // After a rejoin the API answers before p2p info does. + let deadline = Instant::now() + READY_TIMEOUT; + let peer_id = loop { + if let Some(id) = crate::churn::peer_id(&self.http, &url).await { + break id; + } + ensure!( + Instant::now() < deadline, + "{name}: no peer id from {url} within {READY_TIMEOUT:?}" + ); + tokio::time::sleep(Duration::from_millis(500)).await; + }; + let c = &mut self.containers[i]; + c.peer_addr = Some(format!("/ip4/{ip}/tcp/{P2P_PORT}/p2p/{peer_id}")); + c.ip = Some(ip); + Ok(()) + } + + /// Append the container's output since the last dump to + /// `rootdir/logs/{stdout,stderr}.log`. + pub async fn dump_logs(&mut self, i: usize) -> Result<()> { + use std::io::Write; + let name = self.container_name(i); + let since = self.last_dump[i].to_string(); + let now = unix_secs(); + let out = tokio::process::Command::new("docker") + .args(["logs", "--since", &since, &name]) + .output() + .await + .wrap_err_with(|| format!("docker logs {name}"))?; + ensure!( + out.status.success(), + "docker logs {name}: {}", + String::from_utf8_lossy(&out.stderr).trim() + ); + let dir = self.containers[i].rootdir.join("logs"); + std::fs::create_dir_all(&dir)?; + for (file, bytes) in [("stdout.log", &out.stdout), ("stderr.log", &out.stderr)] { + std::fs::OpenOptions::new() + .create(true) + .append(true) + .open(dir.join(file))? + .write_all(bytes)?; + } + self.last_dump[i] = now; + Ok(()) + } + + /// Resident memory of every container from one `docker stats` call; + /// `None` where a container has no usable row. + async fn rss_all(&self) -> Vec> { + let names: Vec = (0..self.containers.len()) + .map(|i| self.container_name(i)) + .collect(); + let mut args = vec![ + "stats", + "--no-stream", + "--format", + "{{.Name}} {{.MemUsage}}", + ]; + args.extend(names.iter().map(String::as_str)); + let out = docker(&args).await.unwrap_or_default(); + let by_name: HashMap<&str, &str> = out.lines().filter_map(|l| l.split_once(' ')).collect(); + names + .iter() + .map(|n| by_name.get(n.as_str()).and_then(|s| parse_mem_usage(s))) + .collect() + } +} + +/// One value per run, so the size gap between the variants is moot. +#[allow(clippy::large_enum_variant)] +pub enum Nodes { + Process { + cluster: TestCluster, + stopped: HashMap, + /// Kind and store per node index, mirroring what Docker keeps on its + /// containers; the process cluster itself does not record them. + specs: Vec, + }, + Docker(DockerNodes), +} + +impl Nodes { + pub fn len(&self) -> usize { + match self { + Nodes::Process { cluster, .. } => cluster.len(), + Nodes::Docker(d) => d.containers.len(), + } + } + + pub fn name(&self, i: usize) -> &str { + match self { + Nodes::Process { cluster, .. } => &cluster.nodes[i].name, + Nodes::Docker(d) => &d.containers[i].spec.name, + } + } + + pub fn api_url(&self, i: usize) -> String { + match self { + Nodes::Process { cluster, .. } => cluster.api_url(i).to_string(), + Nodes::Docker(d) => d.containers[i].api_url(), + } + } + + pub fn store(&self, i: usize) -> &str { + match self { + Nodes::Process { specs, .. } => &specs[i].store, + Nodes::Docker(d) => &d.containers[i].spec.store, + } + } + + pub fn kind(&self, i: usize) -> NodeKind { + match self { + Nodes::Process { specs, .. } => specs[i].kind, + Nodes::Docker(d) => d.containers[i].spec.kind, + } + } + + /// Lowest index running `kind`, or `None` in a single-runtime mesh. + pub fn first_of(&self, kind: NodeKind) -> Option { + (0..self.len()).find(|&i| self.kind(i) == kind) + } + + pub fn rootdir(&self, i: usize) -> PathBuf { + match self { + Nodes::Process { cluster, .. } => cluster.nodes[i].rootdir.clone(), + Nodes::Docker(d) => d.containers[i].rootdir.clone(), + } + } + + /// Host pid of the node process; `None` for a container. + pub fn pid(&self, i: usize) -> Option { + match self { + Nodes::Process { cluster, .. } => cluster.nodes[i].process.id(), + Nodes::Docker(_) => None, + } + } + + /// The container behind node `i`; `None` for a process. + pub fn container(&self, i: usize) -> Option<&Container> { + match self { + Nodes::Process { .. } => None, + Nodes::Docker(d) => Some(&d.containers[i]), + } + } + + /// Resident memory per node: `ps` per process, one `docker stats` call + /// for all containers. + pub async fn rss_all(&self) -> Vec> { + match self { + Nodes::Process { .. } => (0..self.len()) + .map(|i| self.pid(i).and_then(crate::meter::rss_bytes)) + .collect(), + Nodes::Docker(d) => d.rss_all().await, + } + } + + /// Host-side binary per node index, for CLI-only ops. + pub fn binaries(&self) -> Result> { + (0..self.len()) + .map(|i| self.kind(i).host_binary()) + .collect() + } + + pub fn client(&self, i: usize) -> DefraClient { + match self { + Nodes::Process { cluster, .. } => cluster.client(i), + Nodes::Docker(d) => { + let c = &d.containers[i]; + let bin = c + .spec + .kind + .host_binary() + .expect("host binary for the docker client"); + DefraClient::new( + bin, + format!("127.0.0.1:{}", c.api_port), + c.spec.kind.harness(), + ) + } + } + } + + /// The node's dialable libp2p address. + pub fn p2p_addr(&self, i: usize) -> String { + match self { + Nodes::Process { cluster, .. } => extract_p2p_addr(cluster, i), + Nodes::Docker(d) => d.containers[i] + .peer_addr + .clone() + .expect("peer address is recorded at start"), + } + } + + pub fn log_dir(&self, i: usize) -> PathBuf { + match self { + Nodes::Process { cluster, .. } => cluster.nodes[i].process.log_dir().to_path_buf(), + Nodes::Docker(d) => d.containers[i].rootdir.join("logs"), + } + } + + /// Flush container output to `log_dir`; the harness already writes + /// process logs there. + pub async fn dump_logs(&mut self, i: usize) -> Result<()> { + match self { + Nodes::Process { .. } => Ok(()), + Nodes::Docker(d) => d.dump_logs(i).await, + } + } + + pub async fn restart(&mut self, i: usize) -> Result<()> { + match self { + Nodes::Process { cluster, .. } => { + cluster.restart_node(i, Duration::from_secs(60)).await + } + Nodes::Docker(d) => { + docker(&["restart", "-t", STOP_TIMEOUT, &d.container_name(i)]).await?; + Ok(()) + } + } + } + + pub async fn kill(&mut self, i: usize) -> Result<()> { + match self { + Nodes::Process { cluster, .. } => { + cluster.nodes[i].process.kill(); + Ok(()) + } + Nodes::Docker(d) => { + docker(&["kill", &d.container_name(i)]).await?; + Ok(()) + } + } + } + + pub async fn respawn(&mut self, i: usize) -> Result<()> { + match self { + Nodes::Process { cluster, .. } => cluster.nodes[i].process.respawn(), + Nodes::Docker(d) => { + docker(&["start", &d.container_name(i)]).await?; + Ok(()) + } + } + } + + pub async fn stop(&mut self, i: usize) -> Result<()> { + match self { + Nodes::Process { + cluster, stopped, .. + } => { + let node = cluster.stop_node(i).await?; + stopped.insert(i, node); + Ok(()) + } + Nodes::Docker(d) => { + docker(&["stop", "-t", STOP_TIMEOUT, &d.container_name(i)]).await?; + Ok(()) + } + } + } + + pub async fn start_stopped(&mut self, i: usize) -> Result<()> { + match self { + Nodes::Process { + cluster, stopped, .. + } => { + let Some(node) = stopped.remove(&i) else { + bail!("node {i} is not stopped"); + }; + cluster + .start_stopped_node(node, Duration::from_secs(60)) + .await + } + Nodes::Docker(d) => { + docker(&["start", &d.container_name(i)]).await?; + Ok(()) + } + } + } + + pub fn supports_partition(&self) -> bool { + matches!(self, Nodes::Docker(_)) + } + + /// Cut the node off the soak network. Docker unpublishes the API port + /// with the network, so the driver cannot reach the node either: from + /// the driver's side this is a crash-kill whose process keeps running. + /// The port returns on `rejoin`, which re-reads the container's address. + pub async fn partition(&mut self, i: usize) -> Result<()> { + match self { + Nodes::Process { .. } => bail!("process backend cannot partition"), + Nodes::Docker(d) => { + let name = d.container_name(i); + docker(&["network", "disconnect", &d.network, &name]).await?; + Ok(()) + } + } + } + + pub async fn rejoin(&mut self, i: usize) -> Result<()> { + match self { + Nodes::Process { .. } => bail!("process backend cannot partition"), + Nodes::Docker(d) => { + let name = d.container_name(i); + let Some(ip) = d.containers[i].ip.clone() else { + bail!("{name}: no recorded address to rejoin at"); + }; + docker(&["network", "connect", "--ip", &ip, &d.network, &name]).await?; + d.record_addr(i).await + } + } + } + + pub async fn shutdown(self) -> Result<()> { + match self { + Nodes::Process { .. } => Ok(()), + Nodes::Docker(mut d) => d.teardown().await, + } + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn docker_run_argv_shape() { + let c = Container { + spec: NodeSpec { + name: "rust-0".into(), + kind: NodeKind::Rust, + store: "regolith".into(), + host: "A", + }, + image: "soak-defra:8d8bb299f".into(), + api_port: 41181, + rootdir: "/tmp/r0".into(), + ip: None, + peer_addr: None, + }; + let v = docker_run_argv(&c, "soak-test", "s3cret", None, &[]); + let s = v.join(" "); + assert!(s.starts_with("run -d --name soak-test-rust-0 --network soak-test -p 127.0.0.1:41181:9181 -v /tmp/r0:/data -e DEFRA_KEYRING_SECRET=s3cret soak-defra:8d8bb299f ")); + assert!(s.contains("--rootdir /data --url 0.0.0.0:9181 --no-log-color --keyring-backend file --keyring-path /data/keys start --store regolith --no-telemetry --no-encryption --no-searchable-encryption --no-signing --p2paddr /ip4/0.0.0.0/tcp/9171")); + let g = Container { + spec: NodeSpec { + name: "go-0".into(), + kind: NodeKind::Go, + store: "badger".into(), + host: "A", + }, + image: "soak-defradb:53f0e76a3".into(), + ..c + }; + assert!(docker_run_argv(&g, "soak-test", "s", None, &[]).join(" ").contains("soak-defradb:53f0e76a3 --rootdir /data --no-log-color --keyring-backend file --keyring-path /data/keys start --url 0.0.0.0:9181 --store badger --no-telemetry --no-encryption --no-searchable-encryption --no-signing --p2paddr /ip4/0.0.0.0/tcp/9171")); + } + + /// Both runtimes take `--replicator-retry-intervals` on `start` + /// (Rust `crates/cli/src/commands/start/mod.rs`, Go `cli/start.go`), so + /// the container churn can be measured off the default ladder. + #[test] + fn docker_run_argv_carries_the_retry_ladder_for_both_runtimes() { + let rust = Container { + spec: NodeSpec { + name: "rust-0".into(), + kind: NodeKind::Rust, + store: "regolith".into(), + host: "A", + }, + image: "soak-defra:8d8bb299f".into(), + api_port: 41181, + rootdir: "/tmp/r0".into(), + ip: None, + peer_addr: None, + }; + let go = Container { + spec: NodeSpec { + name: "go-0".into(), + kind: NodeKind::Go, + store: "badger".into(), + host: "A", + }, + image: "soak-defradb:53f0e76a3".into(), + ..rust.clone() + }; + for c in [&rust, &go] { + let v = docker_run_argv(c, "soak-test", "s", Some("5,10,20,40"), &[]); + let start = v.iter().position(|a| a == "start").expect("start verb"); + let flag = v + .iter() + .position(|a| a == "--replicator-retry-intervals") + .expect("retry ladder flag"); + // Both runtimes take it on the `start` subcommand, not before it. + assert!(flag > start); + assert_eq!(v[flag + 1], "5,10,20,40"); + // Nothing else moved. + assert!(v.join(" ").contains("--p2paddr /ip4/0.0.0.0/tcp/9171")); + } + // Absent by default, so an unflagged run is byte-identical to before. + assert!(!docker_run_argv(&rust, "soak-test", "s", None, &[]) + .contains(&"--replicator-retry-intervals".to_string())); + } + + /// `--node-env` reaches every container, both runtimes, as `-e KEY=VALUE` + /// before the image, and nothing is added when it is empty. + #[test] + fn docker_run_argv_carries_node_env_for_both_runtimes() { + let rust = Container { + spec: NodeSpec { + name: "rust-0".into(), + kind: NodeKind::Rust, + store: "regolith".into(), + host: "A", + }, + image: "soak-defra:8d8bb299f".into(), + api_port: 41181, + rootdir: "/tmp/r0".into(), + ip: None, + peer_addr: None, + }; + let go = Container { + spec: NodeSpec { + name: "go-0".into(), + kind: NodeKind::Go, + store: "badger".into(), + host: "A", + }, + image: "soak-defradb:53f0e76a3".into(), + ..rust.clone() + }; + let env = [ + "RUST_LOG=debug".to_string(), + "GOLOG_LEVEL=debug".to_string(), + ]; + for c in [&rust, &go] { + let v = docker_run_argv(c, "soak-test", "s", None, &env); + let image = v.iter().position(|a| *a == c.image).expect("image"); + for kv in &env { + let at = v.iter().position(|a| a == kv).expect("node env value"); + assert_eq!(v[at - 1], "-e"); + // Docker only reads `-e` before the image name. + assert!(at < image); + } + // The keyring secret is still there, and the command still starts. + assert!(v.contains(&format!("DEFRA_KEYRING_SECRET={}", "s"))); + assert!(v.contains(&"start".to_string())); + } + // Empty list: byte-identical to a run without the flag. + assert_eq!( + docker_run_argv(&rust, "soak-test", "s", None, &[]), + docker_run_argv(&rust, "soak-test", "s", None, &Vec::new()) + ); + assert!(!docker_run_argv(&rust, "soak-test", "s", None, &[]) + .contains(&"RUST_LOG=debug".to_string())); + } + + #[test] + fn mem_usage_parses_docker_stats() { + assert_eq!( + parse_mem_usage("12.5MiB / 7.7GiB"), + Some((12.5 * 1024.0 * 1024.0) as u64) + ); + assert_eq!( + parse_mem_usage("1.2GiB / 7.7GiB"), + Some((1.2 * 1024.0 * 1024.0 * 1024.0) as u64) + ); + assert_eq!(parse_mem_usage("900kB / 1GB"), Some(900_000)); + assert_eq!(parse_mem_usage("garbage"), None); + } + + #[test] + fn topology_specs_are_rust_first_and_balanced() { + let s = topology_specs(2, 2); + let names: Vec<&str> = s.iter().map(|n| n.name.as_str()).collect(); + assert_eq!(names, ["rust-0", "rust-1", "go-0", "go-1"]); + assert!(s[..2].iter().all(|n| matches!(n.kind, NodeKind::Rust))); + assert!(s[2..].iter().all(|n| matches!(n.kind, NodeKind::Go))); + assert_eq!(s[0].store, "regolith"); + assert_eq!(s[2].store, "badger"); + // Round-robin within each kind, so both groups hold both runtimes. + assert_eq!(s[0].host, "A"); + assert_eq!(s[1].host, "B"); + assert_eq!(s[2].host, "A"); + assert_eq!(s[3].host, "B"); + } + + #[test] + fn topology_specs_allow_a_single_runtime() { + let r = topology_specs(4, 0); + assert_eq!(r.len(), 4); + assert!(r.iter().all(|n| matches!(n.kind, NodeKind::Rust))); + assert!(r.iter().all(|n| n.store == "regolith")); + + let g = topology_specs(0, 4); + assert_eq!(g.len(), 4); + assert!(g.iter().all(|n| matches!(n.kind, NodeKind::Go))); + assert!(g.iter().all(|n| n.store == "badger")); + assert_eq!(g[0].name, "go-0"); + } + + #[test] + fn topology_specs_balance_groups_at_any_size() { + for (r, g) in [(1, 0), (3, 1), (5, 5), (0, 7)] { + let s = topology_specs(r, g); + assert_eq!(s.len(), r + g); + let a = s.iter().filter(|n| n.host == "A").count(); + let b = s.iter().filter(|n| n.host == "B").count(); + assert!(a.abs_diff(b) <= 2, "unbalanced {r}r{g}g: {a} vs {b}"); + } + } + + #[test] + fn m2_specs_split_hosts() { + let s = m2_specs(); + assert_eq!(s.len(), 6); + assert_eq!(s.iter().filter(|n| n.host == "A").count(), 3); + assert_eq!( + s.iter() + .filter(|n| matches!(n.kind, NodeKind::Rust)) + .count(), + 3 + ); + assert_eq!( + s.iter().map(|n| n.name.as_str()).collect::>(), + ["rust-0", "rust-1", "go-0", "rust-2", "go-1", "go-2"] + ); + } +} diff --git a/crates/soak/src/sse.rs b/crates/soak/src/sse.rs new file mode 100644 index 0000000..297bb38 --- /dev/null +++ b/crates/soak/src/sse.rs @@ -0,0 +1,163 @@ +//! GraphQL subscriptions over server-sent events, one per node. +//! +//! Both runtimes answer `POST /api/v0/graphql` with `Accept: +//! text/event-stream` for a `subscription { ... }` document. Each event's +//! `data:` payload is a GraphQL response carrying the docs that changed on +//! that node, including remote merges. The checker uses the arrivals for +//! lag samples with sub-second resolution and as its quiescence trigger. + +use std::time::Duration; + +use futures::StreamExt; +use serde_json::{json, Value}; +use tokio::sync::mpsc; + +use crate::executor::now_ms; + +/// A doc changed on `node`, per its subscription stream. +#[derive(Debug)] +pub struct Arrival { + pub node: usize, + pub doc_id: String, + pub wall_ts_ms: u64, +} + +/// Incremental server-sent-events parser: feed body chunks, get back the +/// `data` payload of every completed event. Multi-line `data:` fields are +/// joined with newlines; comment lines and other fields are ignored. +#[derive(Default)] +pub struct SseParser { + buffer: String, + data: Vec, +} + +impl SseParser { + pub fn push(&mut self, chunk: &[u8]) -> Vec { + self.buffer.push_str(&String::from_utf8_lossy(chunk)); + let mut out = Vec::new(); + while let Some(pos) = self.buffer.find('\n') { + let line: String = self.buffer.drain(..=pos).collect(); + let line = line.trim_end_matches(['\n', '\r']); + if line.is_empty() { + if !self.data.is_empty() { + out.push(std::mem::take(&mut self.data).join("\n")); + } + } else if let Some(rest) = line.strip_prefix("data:") { + self.data + .push(rest.strip_prefix(' ').unwrap_or(rest).to_string()); + } + } + out + } +} + +/// Every `_docID` string anywhere in a GraphQL payload. +fn doc_ids(v: &Value, out: &mut Vec) { + match v { + Value::Object(o) => { + if let Some(id) = o.get("_docID").and_then(Value::as_str) { + out.push(id.to_string()); + } + o.values().for_each(|x| doc_ids(x, out)); + } + Value::Array(a) => a.iter().for_each(|x| doc_ids(x, out)), + _ => {} + } +} + +/// Keep a subscription open on `url`, reconnecting after any error or end +/// of stream (a restarted node), forwarding every changed doc as an +/// [`Arrival`]. Runs until the receiver is dropped. +pub async fn subscribe( + node: usize, + url: String, + query: String, + tx: mpsc::UnboundedSender, +) { + let http = reqwest::Client::new(); + loop { + let stream = http + .post(format!("{url}/api/v0/graphql")) + .header("Accept", "text/event-stream") + .json(&json!({ "query": query })) + .send() + .await; + if let Ok(resp) = stream { + let mut parser = SseParser::default(); + let mut body = resp.bytes_stream(); + while let Some(Ok(chunk)) = body.next().await { + for payload in parser.push(&chunk) { + let Ok(v) = serde_json::from_str::(&payload) else { + continue; + }; + let mut ids = Vec::new(); + doc_ids(&v, &mut ids); + let wall_ts_ms = now_ms(); + for doc_id in ids { + if tx + .send(Arrival { + node, + doc_id, + wall_ts_ms, + }) + .is_err() + { + return; + } + } + } + } + } + if tx.is_closed() { + return; + } + tokio::time::sleep(Duration::from_secs(1)).await; + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn one_event() { + let mut p = SseParser::default(); + assert_eq!(p.push(b"data: {\"a\":1}\n\n"), vec!["{\"a\":1}"]); + } + + #[test] + fn two_events_in_one_chunk_and_comments_ignored() { + let mut p = SseParser::default(); + let out = p.push(b": keepalive\ndata: x\n\nevent: next\ndata: y\n\n"); + assert_eq!(out, vec!["x", "y"]); + } + + #[test] + fn event_split_across_chunks() { + let mut p = SseParser::default(); + assert!(p.push(b"data: {\"a\":").is_empty()); + assert!(p.push(b"1}\n").is_empty()); + assert_eq!(p.push(b"\n"), vec!["{\"a\":1}"]); + } + + #[test] + fn multi_line_data_joined() { + let mut p = SseParser::default(); + assert_eq!(p.push(b"data: a\ndata: b\n\n"), vec!["a\nb"]); + } + + #[test] + fn crlf_and_leading_space_variants() { + let mut p = SseParser::default(); + assert_eq!(p.push(b"data:x\r\n\r\ndata: y\r\n\r\n"), vec!["x", "y"]); + } + + #[test] + fn doc_ids_found_anywhere() { + let v = + json!({"data": {"Users": [{"_docID": "a"}, {"_docID": "b", "x": {"_docID": "c"}}]}}); + let mut out = Vec::new(); + doc_ids(&v, &mut out); + assert_eq!(out, vec!["a", "b", "c"]); + } +} diff --git a/crates/soak/src/summary.rs b/crates/soak/src/summary.rs new file mode 100644 index 0000000..eb57743 --- /dev/null +++ b/crates/soak/src/summary.rs @@ -0,0 +1,1167 @@ +//! End-of-run profile: reads the artifact's JSONL files back and writes +//! `profile.json` plus a one-screen `profile.md` (`soak summarize DIR`), and +//! the replay-contract comparison of two runs (`soak compare A B`). + +use std::collections::{BTreeMap, HashMap, HashSet}; +use std::fs; +use std::path::Path; + +use eyre::{Result, WrapErr}; +use serde_json::{json, Value}; + +use crate::meter::percentile; + +fn read_jsonl(path: &Path) -> Vec { + fs::read_to_string(path) + .unwrap_or_default() + .lines() + .filter_map(|l| serde_json::from_str(l).ok()) + .collect() +} + +fn s<'a>(v: &'a Value, key: &str) -> &'a str { + v[key].as_str().unwrap_or("") +} + +fn pct(values: &[u64]) -> Value { + json!({ + "n": values.len(), + "p50": percentile(values, 0.5), + "p95": percentile(values, 0.95), + "max": values.iter().max(), + }) +} + +/// Build the profile from `run_dir` and write `profile.json` / `profile.md`. +pub fn write_profile(run_dir: &Path) -> Result { + let manifest: Value = serde_json::from_str( + &fs::read_to_string(run_dir.join("manifest.json")).wrap_err("reading manifest.json")?, + )?; + let ops = read_jsonl(&run_dir.join("ops.jsonl")); + let du = read_jsonl(&run_dir.join("du.jsonl")); + let rss = read_jsonl(&run_dir.join("rss.jsonl")); + let lag = read_jsonl(&run_dir.join("lag.jsonl")); + let checks = read_jsonl(&run_dir.join("checks.jsonl")); + let divergences = read_jsonl(&run_dir.join("divergences.jsonl")); + let topology = read_jsonl(&run_dir.join("topology.jsonl")); + + // Latency per (node, kind); outcome counts. + let mut latency: BTreeMap<(String, String), Vec> = BTreeMap::new(); + let (mut ok, mut failed, mut skipped, mut writes_ok) = (0u64, 0u64, 0u64, 0u64); + for op in &ops { + let is_ok = op["ok"].as_bool().unwrap_or(false); + if is_ok { + ok += 1; + if is_write(s(op, "kind")) { + writes_ok += 1; + } + latency + .entry((s(op, "node").to_string(), s(op, "kind").to_string())) + .or_default() + .push(op["latency_ms"].as_u64().unwrap_or(0)); + } else if op["skipped"].as_bool().unwrap_or(false) { + skipped += 1; + } else { + failed += 1; + } + } + let span_s = match (ops.first(), ops.last()) { + (Some(a), Some(b)) if ops.len() > 1 => { + (b["wall_ts_ms"].as_f64().unwrap_or(0.0) - a["wall_ts_ms"].as_f64().unwrap_or(0.0)) + / 1000.0 + } + _ => 0.0, + }; + + // Disk: first and last sample per node; bytes per mesh-wide write op. + let mut disk: BTreeMap = BTreeMap::new(); + for d in &du { + let bytes = d["bytes"].as_u64().unwrap_or(0); + disk.entry(s(d, "node").to_string()) + .and_modify(|e| e.1 = bytes) + .or_insert((bytes, bytes)); + } + let mut rss_max: BTreeMap = BTreeMap::new(); + for r in &rss { + let b = r["rss_bytes"].as_u64().unwrap_or(0); + let e = rss_max.entry(s(r, "node").to_string()).or_default(); + *e = (*e).max(b); + } + // Lag in three groupings: directed pair, source alone, and receiving + // runtime x source. `source` is what the sample came from (poll or sse); + // mixing them hides that an sse sample is an event-time arrival. + let mut lag_sources: BTreeMap = BTreeMap::new(); + let mut lag_groups: BTreeMap<(String, String), Vec> = BTreeMap::new(); + let mut lag_seen: BTreeMap<(String, String), HashSet<&str>> = BTreeMap::new(); + for l in &lag { + let source = l["source"].as_str().unwrap_or("poll"); + let ms = l["lag_ms"].as_u64().unwrap_or(0); + *lag_sources.entry(source.to_string()).or_default() += 1; + for key in [ + ( + format!("{}->{}", s(l, "from"), s(l, "to")), + "all".to_string(), + ), + ("all".to_string(), source.to_string()), + (format!("*->{}", runtime(s(l, "to"))), source.to_string()), + ] { + lag_groups.entry(key).or_default().push(ms); + } + if let Some(id) = l["doc_id"].as_str() { + lag_seen + .entry((s(l, "from").to_string(), s(l, "to").to_string())) + .or_default() + .insert(id); + } + } + // Censoring: creates that never produced a lag sample on the far side are + // not fast, they are unseen, and they are absent from every percentile. + let mut creates_by_node: BTreeMap<&str, HashSet<&str>> = BTreeMap::new(); + for op in &ops { + if op["ok"].as_bool() == Some(true) && s(op, "kind") == "create" { + if let Some(id) = op["doc_id"].as_str() { + creates_by_node.entry(s(op, "node")).or_default().insert(id); + } + } + } + let node_names: Vec<&str> = manifest["nodes"] + .as_array() + .into_iter() + .flatten() + .filter_map(|n| n["name"].as_str()) + .collect(); + let mut lag_unseen: Vec = Vec::new(); + for (from, creates) in &creates_by_node { + for to in &node_names { + if to == from { + continue; + } + let seen = lag_seen.get(&(from.to_string(), to.to_string())); + let seen_here = seen.map_or(0, |s| creates.iter().filter(|d| s.contains(*d)).count()); + lag_unseen.push(json!({"from": from, "to": to, "creates": creates.len(), + "lag_samples": seen.map_or(0, HashSet::len), + "unseen": creates.len() - seen_here})); + } + } + let mut check_status: BTreeMap = BTreeMap::new(); + for c in &checks { + *check_status.entry(s(c, "status").to_string()).or_default() += 1; + } + let final_check = checks + .iter() + .rev() + .find(|c| c["full"].as_bool() == Some(true)); + let diverged_docs: usize = divergences + .iter() + .map(|d| d["doc_ids"].as_array().map_or(0, Vec::len)) + .sum(); + let mut record_tags: BTreeMap = BTreeMap::new(); + for d in &divergences { + for t in d["doc_tags"].as_array().into_iter().flatten() { + *record_tags + .entry(t.as_str().unwrap_or("untagged").to_string()) + .or_default() += 1; + } + } + let final_sweep_lines = read_jsonl(&run_dir.join("final_sweep.jsonl")); + let mut sweep_tags: BTreeMap = BTreeMap::new(); + for f in &final_sweep_lines { + *sweep_tags + .entry(f["tag"].as_str().unwrap_or("untagged").to_string()) + .or_default() += 1; + } + let sweep_docs: std::collections::HashSet<&str> = final_sweep_lines + .iter() + .filter_map(|f| f["doc_id"].as_str()) + .collect(); + // Unique documents, not rows: a sweep row is one (pair, doc) mismatch, so a + // doc missing on one node shows up once per pair that node is in. + let mut m1_by_missing: BTreeMap<&str, HashSet<&str>> = BTreeMap::new(); + let mut m1_docs: HashSet<&str> = HashSet::new(); + for f in &final_sweep_lines { + if s(f, "mechanism") != "M1" { + continue; + } + let (Some(doc), Some(node)) = (f["doc_id"].as_str(), f["detail"]["missing_on"].as_str()) + else { + continue; + }; + m1_docs.insert(doc); + m1_by_missing.entry(node).or_default().insert(doc); + } + let union_where = |pred: fn(&str) -> bool| -> HashSet<&str> { + m1_by_missing + .iter() + .filter(|(n, _)| pred(n)) + .flat_map(|(_, d)| d.iter().copied()) + .collect() + }; + let m1_on_rust = union_where(|n| n.starts_with("rust")); + let m1_on_go = union_where(|n| n.starts_with("go")); + let diverged_docs_unique: HashSet<&str> = divergences + .iter() + .flat_map(|d| d["doc_ids"].as_array().into_iter().flatten()) + .filter_map(Value::as_str) + .collect(); + let mut record_docs_healed = 0u64; + let mut record_docs_persistent = 0u64; + for d in &divergences { + for id in d["doc_ids"].as_array().into_iter().flatten() { + if sweep_docs.contains(id.as_str().unwrap_or("")) { + record_docs_persistent += 1; + } else { + record_docs_healed += 1; + } + } + } + let mut records_by_pair: BTreeMap = BTreeMap::new(); + for d in &divergences { + let pair = d["pair"] + .as_array() + .map(|p| { + p.iter() + .filter_map(Value::as_str) + .collect::>() + .join("|") + }) + .unwrap_or_default(); + *records_by_pair.entry(pair).or_default() += 1; + } + let mut churn_kinds: BTreeMap = BTreeMap::new(); + let mut longest_outage_ms = 0u64; + for t in &topology { + if s(t, "phase") == "up" { + *churn_kinds.entry(s(t, "kind").to_string()).or_default() += 1; + longest_outage_ms = longest_outage_ms.max(t["duration_ms"].as_u64().unwrap_or(0)); + } + } + // Docker samples carry a null pid: those bytes are `docker stats` MemUsage + // for the whole container, not the process RSS `ps` reports. + let rss_instrument = if !rss.is_empty() && rss.iter().all(|r| r["pid"].is_null()) { + "docker_stats" + } else { + "ps" + }; + let causes = classify_final_sweep( + &read_jsonl(&run_dir.join("final_sweep.jsonl")), + &ops, + &topology, + ); + let stores: HashMap = manifest["nodes"] + .as_array() + .into_iter() + .flatten() + .map(|n| (s(n, "name").to_string(), s(n, "store").to_string())) + .collect(); + + let profile = json!({ + "run_id": manifest["run_id"], + "seed": manifest["seed"], + "ops": {"executed": ops.len(), "ok": ok, "failed": failed, "skipped": skipped, + "writes_ok": writes_ok, "span_s": span_s, + "achieved_ops_per_s": if span_s > 0.0 { (ops.len() as f64 - 1.0) / span_s } else { 0.0 }, + "stopped_by": manifest["stopped_by"]}, + "latency_ms": latency.iter().map(|((node, kind), v)| { + json!({"node": node, "kind": kind, "stats": pct(v)}) + }).collect::>(), + "disk": disk.iter().map(|(node, (first, last))| json!({ + "node": node, "store": stores.get(node), "start_bytes": first, "end_bytes": last, + "bytes_per_write_op": if writes_ok > 0 { (last.saturating_sub(*first)) as f64 / writes_ok as f64 } else { 0.0 }, + })).collect::>(), + "payload_bytes": payload_summary(&ops), + "disk_fit": disk_fit(&ops, &du), + "rss_max_bytes": rss_max, + "rss_instrument": rss_instrument, + "convergence_lag_ms": lag_groups.iter().map(|((dir, source), v)| json!({"direction": dir, "source": source, "stats": pct(v)})).collect::>(), + "lag_unseen": lag_unseen, + "lag_sources": lag_sources, + "checks": check_status, + "divergence_records": divergences.len(), + "diverged_docs": diverged_docs, + "diverged_docs_unique": diverged_docs_unique.len(), + "records_by_pair": records_by_pair, + "record_doc_tags": record_tags.clone(), + "record_doc_tag_slots": record_tags, + "record_docs_healed_by_sweep": record_docs_healed, + "record_docs_persistent": record_docs_persistent, + "final_sweep_tags": sweep_tags, + "final_sweep": final_check.map(|c| json!({"mismatches": c["mismatches"], "eligible": c["eligible"], + "sampled_pending": c["pending"]})), + "sweep_rows": final_sweep_lines.len(), + "sweep_unique_docs": sweep_docs.len(), + "sweep_unique_m1_docs": m1_docs.len(), + "sweep_unique_m1_by_missing_on": m1_by_missing.iter().map(|(n, d)| (n.to_string(), d.len())).collect::>(), + "sweep_unique_m1_on_rust": m1_on_rust.len(), + "sweep_unique_m1_on_go": m1_on_go.len(), + "sweep_unique_m1_overlap": m1_on_rust.intersection(&m1_on_go).count(), + "loss_strict": loss_by_outage(&ops, &topology, &final_sweep_lines, 0), + "loss_plus_30s": loss_by_outage(&ops, &topology, &final_sweep_lines, RECOVERY_WINDOW_MS), + "final_sweep_causes": causes, + "churn": {"events": churn_kinds, "longest_outage_ms": longest_outage_ms}, + }); + fs::write( + run_dir.join("profile.json"), + serde_json::to_string_pretty(&profile)?, + )?; + fs::write(run_dir.join("profile.md"), render(&profile, &manifest))?; + Ok(profile) +} + +fn mb(v: &Value) -> String { + format!("{:.1}", v.as_f64().unwrap_or(0.0) / 1_048_576.0) +} + +/// One kind's payload-size line. A run predating `payload_bytes` has `sum == +/// 0` for every op that carried a real payload; printing that as `sum 0 p50 +/// 0 p95 0` reads as "no bytes", when the true figure was just never +/// recorded, so such a kind renders `not recorded` instead of the numbers. +fn payload_line(p: &Value, kind: &str) -> String { + let field = &p["payload_bytes"][kind]; + let n = field["n"].as_u64().unwrap_or(0); + if n > 0 && field["sum"].as_u64().unwrap_or(0) == 0 { + return format!("{kind} payload: n {n} not recorded"); + } + if kind == "create" { + format!( + "{kind} payload: n {n} sum {} p50 {} p95 {}", + field["sum"], field["p50"], field["p95"], + ) + } else { + format!("{kind} payload: n {n} p50 {}", field["p50"]) + } +} + +fn render(p: &Value, manifest: &Value) -> String { + let mut out = String::new(); + let o = &p["ops"]; + out += &format!( + "# soak profile: run {} (seed {})\n\nrust {} / go {}\n\nops: {} executed ({} ok, {} failed, {} skipped), {:.2} ops/s over {:.0}s, {} mesh writes ok, stopped by {}\n\n", + p["run_id"].as_str().unwrap_or("?"), + p["seed"], + manifest["rust_version"]["commit"].as_str().map(|c| &c[..c.len().min(9)]).unwrap_or("?"), + manifest["go_version"]["commit"].as_str().map(|c| &c[..c.len().min(9)]).unwrap_or("?"), + o["executed"], o["ok"], o["failed"], o["skipped"], + o["achieved_ops_per_s"].as_f64().unwrap_or(0.0), + o["span_s"].as_f64().unwrap_or(0.0), + o["writes_ok"], + o["stopped_by"].as_str().unwrap_or("?"), + ); + out += "## latency ms (p50 / p95 / max, n)\n\n| node | kind | p50 | p95 | max | n |\n|---|---|---|---|---|---|\n"; + for l in p["latency_ms"].as_array().into_iter().flatten() { + let st = &l["stats"]; + out += &format!( + "| {} | {} | {} | {} | {} | {} |\n", + l["node"].as_str().unwrap_or(""), + l["kind"].as_str().unwrap_or(""), + st["p50"], + st["p95"], + st["max"], + st["n"] + ); + } + out += "\n## disk (engine-inclusive: store named per node)\n\n| node | store | start MB | end MB | bytes_grown_per_mesh_write |\n|---|---|---|---|---|\n"; + for d in p["disk"].as_array().into_iter().flatten() { + out += &format!( + "| {} | {} | {} | {} | {:.0} |\n", + d["node"].as_str().unwrap_or(""), + d["store"].as_str().unwrap_or("?"), + mb(&d["start_bytes"]), + mb(&d["end_bytes"]), + d["bytes_per_write_op"].as_f64().unwrap_or(0.0) + ); + } + out += &format!( + "\n{} ยท {}\n", + payload_line(p, "create"), + payload_line(p, "update"), + ); + out += "\ndisk fit (per node, mesh-wide write/payload regressors against that node's own disk series):\n"; + for f in p["disk_fit"].as_array().into_iter().flatten() { + if f["fitted"].as_bool() == Some(true) { + out += &format!( + "- {}: {:.0} bytes per write plus {:.2}x payload over {} samples\n", + f["node"].as_str().unwrap_or(""), + f["overhead_bytes_per_write"].as_f64().unwrap_or(0.0), + f["amplification_per_payload_byte"].as_f64().unwrap_or(0.0), + f["n_samples"], + ); + } else { + out += &format!( + "- {}: not fitted ({})\n", + f["node"].as_str().unwrap_or(""), + f["reason"].as_str().unwrap_or("") + ); + } + } + out += if p["rss_instrument"] == "docker_stats" { + "\n## max memory (docker stats MemUsage, not process RSS) MB\n\n" + } else { + "\n## max RSS MB (ps rss)\n\n" + }; + for (node, b) in p["rss_max_bytes"].as_object().into_iter().flatten() { + out += &format!("- {node}: {}\n", mb(b)); + } + out += &format!( + "\n## convergence lag ms (create first seen on the other side; sources {})\n\n| direction | source | n | p50_ms | p95_ms | max_ms |\n|---|---|---|---|---|---|\n", + p["lag_sources"] + ); + for l in p["convergence_lag_ms"].as_array().into_iter().flatten() { + let st = &l["stats"]; + out += &format!( + "| {} | {} | {} | {} | {} | {} |\n", + l["direction"].as_str().unwrap_or(""), + l["source"].as_str().unwrap_or(""), + st["n"], + st["p50"], + st["p95"], + st["max"] + ); + } + let unseen: Vec<&Value> = p["lag_unseen"] + .as_array() + .into_iter() + .flatten() + .filter(|u| u["unseen"].as_u64().unwrap_or(0) > 0) + .collect(); + if !unseen.is_empty() { + out += "\ncreates with no lag sample on the far side (censored, absent from every percentile above):\n\n"; + for u in unseen { + out += &format!( + "- {} -> {}: {} of {} creates unseen\n", + u["from"].as_str().unwrap_or(""), + u["to"].as_str().unwrap_or(""), + u["unseen"], + u["creates"] + ); + } + } + out += &format!( + "\n## checks: {}; divergence records {} ({} record-doc slots, {} unique docs) by pair {}; final sweep {}\n\n`sampled_pending` is what the last full pass happened to look at (recent docs plus a cold sample of 50), not the size of the backlog.\n", + p["checks"], + p["divergence_records"], + p["diverged_docs"], + p["diverged_docs_unique"], + p["records_by_pair"], + p["final_sweep"] + ); + out += &format!( + "\nfinal sweep: {} rows, {} unique docs (M1 unique {}, rust {}, go {}, overlap {})\n", + p["sweep_rows"], + p["sweep_unique_docs"], + p["sweep_unique_m1_docs"], + p["sweep_unique_m1_on_rust"], + p["sweep_unique_m1_on_go"], + p["sweep_unique_m1_overlap"] + ); + out += &format!( + "\n## known-cause tags: record-doc slots {} ({} slots healed by the sweep, {} still present); final sweep {}\n", + p["record_doc_tags"], p["record_docs_healed_by_sweep"], p["record_docs_persistent"], p["final_sweep_tags"] + ); + if let Some(causes) = p["final_sweep_causes"] + .as_object() + .filter(|c| !c.is_empty()) + { + out += "\n## final sweep mismatches by likely cause (last write on the doc vs the peer's outage windows)\n\n"; + for (cause, n) in causes { + out += &format!("- {n} {cause}\n"); + } + } + for (key, title) in [ + ("loss_strict", "[down,up]"), + ("loss_plus_30s", "[down,up+30s]"), + ] { + let rows = p[key].as_array().into_iter().flatten(); + let mut collapsed: BTreeMap<(&str, &str), (u64, u64)> = BTreeMap::new(); + for r in rows { + let e = collapsed + .entry(( + r["kind"].as_str().unwrap_or(""), + r["recv_rt"].as_str().unwrap_or(""), + )) + .or_default(); + e.0 += r["written"].as_u64().unwrap_or(0); + e.1 += r["lost"].as_u64().unwrap_or(0); + } + if collapsed.is_empty() { + continue; + } + out += &format!( + "\n## loss (creates by another node in {title}, still missing_on the down node at the final sweep)\n\n| kind | recv | written | lost |\n|---|---|---|---|\n" + ); + for ((kind, recv), (written, lost)) in collapsed { + out += &format!("| {kind} | {recv} | {written} | {lost} |\n"); + } + } + out += &format!( + "\n## churn: {} events, longest outage {:.1}s\n", + p["churn"]["events"], + p["churn"]["longest_outage_ms"].as_f64().unwrap_or(0.0) / 1000.0 + ); + out +} + +/// The replay contract: planned op fields (index, virtual time, node, kind, +/// collection) and the churn schedule must match; docIDs must agree wherever +/// both runs learned one. Outcomes, error text and wall timing may differ. +pub fn compare(a: &Path, b: &Path) -> Result { + let planned = |op: &Value| -> (u64, u64, String, String, String) { + ( + op["op_index"].as_u64().unwrap_or(0), + op["virtual_ts_ms"].as_u64().unwrap_or(0), + s(op, "node").to_string(), + s(op, "kind").to_string(), + s(op, "collection").to_string(), + ) + }; + let ops_a = read_jsonl(&a.join("ops.jsonl")); + let ops_b = read_jsonl(&b.join("ops.jsonl")); + let mut planned_diffs = Vec::new(); + let (mut doc_ids_agree, mut doc_ids_differ) = (0usize, 0usize); + for (x, y) in ops_a.iter().zip(&ops_b) { + if planned(x) != planned(y) { + planned_diffs.push(x["op_index"].as_u64().unwrap_or(0)); + } + match (x["doc_id"].as_str(), y["doc_id"].as_str()) { + (Some(p), Some(q)) if p == q => doc_ids_agree += 1, + (Some(_), Some(_)) => doc_ids_differ += 1, + _ => {} + } + } + let schedule = |dir: &Path| -> Value { + let m: Value = fs::read_to_string(dir.join("manifest.json")) + .ok() + .and_then(|t| serde_json::from_str(&t).ok()) + .unwrap_or(Value::Null); + m["churn"]["schedule"].clone() + }; + let (sched_a, sched_b) = (schedule(a), schedule(b)); + let contract = planned_diffs.is_empty() && doc_ids_differ == 0 && sched_a == sched_b; + let identical = contract && ops_a.len() == ops_b.len(); + // A `--until-op` replay is a prefix of the original. + let prefix = contract && ops_b.len() < ops_a.len(); + Ok(json!({ + "identical": identical, + "prefix_identical": prefix, + "ops": {"a": ops_a.len(), "b": ops_b.len(), "planned_diffs": planned_diffs.len(), + "first_planned_diffs": planned_diffs.iter().take(5).collect::>(), + "doc_ids_agree": doc_ids_agree, "doc_ids_differ": doc_ids_differ}, + "churn_schedule_identical": sched_a == sched_b, + })) +} + +/// Writes this soon after either node came back count as made during the +/// recovery: the node answers GraphQL before its replicator link is back. +const RECOVERY_WINDOW_MS: u64 = 30_000; + +/// For each final-sweep mismatch: what was the last successful write to +/// that doc, and was a member of the pair down, or the writer or a member +/// freshly recovered, at that moment? Counts by +/// `" missing_on=: last on while "`. +fn classify_final_sweep( + sweep: &[Value], + ops: &[Value], + topology: &[Value], +) -> BTreeMap { + // Down windows per node: (start wall ms, end wall ms, kind). + let mut ups: HashMap = HashMap::new(); + for t in topology { + if s(t, "phase") == "up" { + ups.insert( + t["event"].as_u64().unwrap_or(0), + t["wall_ts_ms"].as_u64().unwrap_or(0), + ); + } + } + let windows: Vec<(String, u64, u64, String)> = topology + .iter() + .filter(|t| s(t, "phase") == "down") + .map(|t| { + let ev = t["event"].as_u64().unwrap_or(0); + ( + s(t, "node").to_string(), + t["wall_ts_ms"].as_u64().unwrap_or(0), + ups.get(&ev).copied().unwrap_or(u64::MAX), + s(t, "kind").to_string(), + ) + }) + .collect(); + let mut last_write: HashMap<&str, &Value> = HashMap::new(); + for op in ops { + if op["ok"].as_bool() == Some(true) && is_write(s(op, "kind")) { + if let Some(id) = op["doc_id"].as_str() { + last_write.insert(id, op); + } + } + } + let mut out = BTreeMap::new(); + for f in sweep { + let mech = s(f, "mechanism"); + let missing_on = f["detail"]["missing_on"] + .as_str() + .or_else(|| f["detail"]["undecryptable_on"].as_str()) + .map(String::from) + .or_else(|| { + f["detail"]["viewer"] + .as_str() + .map(|v| format!("viewer={v}")) + }) + .unwrap_or_else(|| "-".to_string()); + let members: Vec<&str> = s(f, "pair").split('|').collect(); + let label = match last_write.get(s(f, "doc_id")) { + None => format!("{mech} missing_on={missing_on}: no successful write on record"), + Some(op) => { + let node = s(op, "node"); + let wall = op["wall_ts_ms"].as_u64().unwrap_or(0); + let involved = |n: &str| n == node || members.contains(&n); + let state = if let Some((_, _, _, kind)) = windows + .iter() + .find(|(n, a, b, _)| n != node && involved(n) && *a <= wall && wall <= *b) + { + format!("peer {kind}") + } else if let Some((n, _, _, _)) = windows.iter().find(|(n, _, b, _)| { + involved(n) && *b <= wall && wall - *b <= RECOVERY_WINDOW_MS + }) { + if n == node { + "writer recovering (<30s up)".to_string() + } else { + "peer recovering (<30s up)".to_string() + } + } else { + "both up".to_string() + }; + format!( + "{mech} missing_on={missing_on}: last {} on {node} while {state}", + s(op, "kind") + ) + } + }; + *out.entry(label).or_default() += 1; + } + out +} + +/// A node's runtime is its name prefix; the cluster builder names them. +fn runtime(node: &str) -> &'static str { + if node.starts_with("rust") { + "rust" + } else if node.starts_with("go") { + "go" + } else { + "?" + } +} + +/// Loss during an outage: creates made by some *other* node while a node was +/// down, counted against the ones that node was still missing at the final +/// sweep. `extra` widens the window past the `up` record (a node answers +/// GraphQL before its replicator link is back). Keyed by outage kind, the +/// runtime of the node that was down, and the runtime of the writer. +fn loss_by_outage(ops: &[Value], topology: &[Value], sweep: &[Value], extra: u64) -> Vec { + let ups: HashMap = topology + .iter() + .filter(|t| s(t, "phase") == "up") + .map(|t| { + ( + t["event"].as_u64().unwrap_or(0), + t["wall_ts_ms"].as_u64().unwrap_or(0), + ) + }) + .collect(); + let windows: Vec<(&str, &str, u64, u64)> = topology + .iter() + .filter(|t| s(t, "phase") == "down") + .filter_map(|t| { + let up = ups.get(&t["event"].as_u64().unwrap_or(0))?; + Some(( + s(t, "kind"), + s(t, "node"), + t["wall_ts_ms"].as_u64().unwrap_or(0), + up + extra, + )) + }) + .collect(); + let mut missing_on: HashMap<&str, HashSet<&str>> = HashMap::new(); + for f in sweep { + if s(f, "mechanism") == "M1" { + if let (Some(doc), Some(node)) = + (f["doc_id"].as_str(), f["detail"]["missing_on"].as_str()) + { + missing_on.entry(node).or_default().insert(doc); + } + } + } + let mut agg: BTreeMap<(&str, &str, &str), (u64, u64)> = BTreeMap::new(); + for (kind, node, down, up) in &windows { + for op in ops { + if op["ok"].as_bool() != Some(true) + || s(op, "kind") != "create" + || s(op, "node") == *node + { + continue; + } + let Some(doc) = op["doc_id"].as_str() else { + continue; + }; + let wall = op["wall_ts_ms"].as_u64().unwrap_or(0); + if wall < *down || wall > *up { + continue; + } + let e = agg + .entry((kind, runtime(node), runtime(s(op, "node")))) + .or_default(); + e.0 += 1; + if missing_on.get(node).is_some_and(|m| m.contains(doc)) { + e.1 += 1; + } + } + } + agg.into_iter() + .map(|((kind, recv, writer), (written, lost))| { + json!({"kind": kind, "recv_rt": recv, "writer_rt": writer, + "written": written, "lost": lost}) + }) + .collect() +} + +/// Grants change relationships, not documents, so they are not writes. +fn is_write(kind: &str) -> bool { + !matches!(kind, "query" | "grant") +} + +/// Per-kind payload sizes over successful ops. Failed ops are excluded so the +/// figure matches the `writes_ok` denominator every disk number already uses. +fn payload_summary(ops: &[Value]) -> Value { + let mut out = serde_json::Map::new(); + for kind in ["create", "update"] { + let v: Vec = ops + .iter() + .filter(|o| s(o, "kind") == kind && o["ok"].as_bool() == Some(true)) + .map(|o| o["payload_bytes"].as_u64().unwrap_or(0)) + .collect(); + let sum: u64 = v.iter().sum(); + let stats = pct(&v); + out.insert( + kind.to_string(), + json!({"n": v.len(), "sum": sum, "p50": stats["p50"], "p95": stats["p95"]}), + ); + } + Value::Object(out) +} + +/// Two unknowns (overhead, amplification) are algebraically solvable from as +/// few as two disk samples; that leaves no slack to tell a real fit from +/// noise. Require several samples per fitted term before trusting one at +/// all -- this is a floor on statistical power, not the collinearity check +/// below. +const MIN_DISK_SAMPLES: usize = 10; + +/// `det = sxx*syy - sxy^2 = sxx*syy*(1 - r^2)`, so `det / (sxx*syy)` -- the +/// fitted-line "tolerance" -- is the dimensionless `1 - r^2`, unlike the raw +/// determinant, which scales with the fourth power of the run's byte/write/ +/// payload magnitudes and so is never small at production scale even when +/// the regressors are near-perfectly collinear. A tolerance below 0.1 +/// (equivalently VIF = 1/tolerance above 10) is the standard collinearity- +/// diagnostic threshold in regression practice: below it the two regressors +/// track each other too closely for least squares to split their effects. +const MIN_TOLERANCE: f64 = 0.1; + +/// Split disk growth into fixed per-write overhead and per-payload-byte +/// amplification, per node, by least squares over each node's own disk +/// series against the mesh-wide write/payload regressors. A mixed mesh's +/// nodes do not always agree on the fit (a store's on-disk layout is its +/// own), so publishing one node's numbers under an unlabeled key would +/// attribute one runtime's behaviour to the whole run. +/// +/// Refuses to fit a node when payload was never recorded (every run before +/// `payload_bytes` existed), when payload sizes do not vary, when that node +/// has too few disk samples, when the two regressors are too collinear to +/// separate, or when either fitted term comes out negative -- storing +/// payload bytes cannot shrink the store, and per-write overhead cannot be +/// negative, so a negative term means the collinearity gate above was too +/// permissive rather than a value worth publishing. +fn disk_fit(ops: &[Value], du: &[Value]) -> Value { + let mut nodes: Vec = du.iter().map(|d| s(d, "node").to_string()).collect(); + nodes.sort(); + nodes.dedup(); + + let creates: Vec = ops + .iter() + .filter(|o| s(o, "kind") == "create" && o["ok"].as_bool() == Some(true)) + .map(|o| o["payload_bytes"].as_u64().unwrap_or(0)) + .collect(); + let sizes: std::collections::BTreeSet = creates.iter().copied().collect(); + let mesh_reason = if !creates.is_empty() && creates.iter().sum::() == 0 { + Some("payload was never recorded") + } else if sizes.len() < 2 { + Some("payload sizes do not vary") + } else { + None + }; + if let Some(reason) = mesh_reason { + return Value::Array( + nodes + .iter() + .map(|node| json!({"node": node, "fitted": false, "reason": reason})) + .collect(), + ); + } + + let mut writes: Vec<(u64, u64)> = ops + .iter() + .filter(|o| o["ok"].as_bool() == Some(true) && is_write(s(o, "kind"))) + .map(|o| { + ( + o["wall_ts_ms"].as_u64().unwrap_or(0), + o["payload_bytes"].as_u64().unwrap_or(0), + ) + }) + .collect(); + writes.sort_unstable(); + + Value::Array( + nodes + .iter() + .map(|node| disk_fit_for_node(node, &writes, du)) + .collect(), + ) +} + +/// One node's fit against the mesh-wide `writes` regressors; see `disk_fit` +/// for what each gate refuses and why. +fn disk_fit_for_node(node: &str, writes: &[(u64, u64)], du: &[Value]) -> Value { + let mut series: Vec<(u64, u64)> = du + .iter() + .filter(|d| s(d, "node") == node) + .map(|d| { + ( + d["wall_ts_ms"].as_u64().unwrap_or(0), + d["bytes"].as_u64().unwrap_or(0), + ) + }) + .collect(); + series.sort_unstable(); + if series.len() < MIN_DISK_SAMPLES { + return json!({"node": node, "fitted": false, "reason": format!("fewer than {MIN_DISK_SAMPLES} disk samples")}); + } + let base = series[0].1; + // Cumulative writes/payload as of `t`. Regressors are taken relative to + // series[0]'s own cumulative counts (not zero), because `base` already + // absorbs whatever growth those counts caused; without this the terms + // are biased by however much had already been written before the first + // disk sample. + let cum_at = |t: u64| -> (f64, f64) { + let (mut w, mut p) = (0f64, 0f64); + for (wt, pay) in writes { + if *wt > t { + break; + } + w += 1.0; + p += *pay as f64; + } + (w, p) + }; + let (w0, p0) = cum_at(series[0].0); + + let (mut sxx, mut sxy, mut syy, mut sxz, mut syz) = (0f64, 0f64, 0f64, 0f64, 0f64); + let mut n = 0usize; + for (t, bytes) in &series { + let (w, p) = cum_at(*t); + let (w, p) = (w - w0, p - p0); + let g = bytes.saturating_sub(base) as f64; + sxx += w * w; + sxy += w * p; + syy += p * p; + sxz += w * g; + syz += p * g; + n += 1; + } + let scale = sxx * syy; + let tolerance = if scale > 0.0 { + 1.0 - (sxy * sxy) / scale + } else { + 0.0 + }; + if tolerance < MIN_TOLERANCE { + return json!({"node": node, "fitted": false, "reason": "regressors are collinear"}); + } + let det = sxx * syy - sxy * sxy; + let overhead = (syy * sxz - sxy * syz) / det; + let amplification = (sxx * syz - sxy * sxz) / det; + if overhead < 0.0 || amplification < 0.0 { + return json!({"node": node, "fitted": false, "reason": "fitted term is negative"}); + } + json!({ + "node": node, + "fitted": true, + "reason": "", + "overhead_bytes_per_write": overhead, + "amplification_per_payload_byte": amplification, + "n_samples": n, + }) +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn loss_counts_only_foreign_creates_inside_the_window() { + let topology = vec![ + json!({"event": 1, "phase": "down", "kind": "crash_kill", "node": "go-0", "wall_ts_ms": 100}), + json!({"event": 1, "phase": "up", "kind": "crash_kill", "node": "go-0", "wall_ts_ms": 200}), + ]; + let ops = vec![ + // Kept, and still missing at the sweep. + json!({"ok": true, "kind": "create", "node": "rust-0", "doc_id": "a", "wall_ts_ms": 150}), + // Kept, arrived. + json!({"ok": true, "kind": "create", "node": "rust-0", "doc_id": "b", "wall_ts_ms": 160}), + // The down node's own create: not counted. + json!({"ok": true, "kind": "create", "node": "go-0", "doc_id": "c", "wall_ts_ms": 170}), + // After the window, inside the +30s recovery window. + json!({"ok": true, "kind": "create", "node": "rust-0", "doc_id": "d", "wall_ts_ms": 210}), + // Not a create. + json!({"ok": true, "kind": "update", "node": "rust-0", "doc_id": "e", "wall_ts_ms": 150}), + // Failed. + json!({"ok": false, "kind": "create", "node": "rust-0", "doc_id": "f", "wall_ts_ms": 150}), + ]; + let sweep = vec![ + json!({"mechanism": "M1", "doc_id": "a", "detail": {"missing_on": "go-0"}}), + // M3 rows have no missing_on and must not join. + json!({"mechanism": "M3", "doc_id": "b", "detail": {"heads_a": "x", "heads_b": "y"}}), + ]; + let strict = loss_by_outage(&ops, &topology, &sweep, 0); + assert_eq!(strict.len(), 1); + assert_eq!(strict[0]["kind"], "crash_kill"); + assert_eq!(strict[0]["recv_rt"], "go"); + assert_eq!(strict[0]["writer_rt"], "rust"); + assert_eq!(strict[0]["written"], 2); + assert_eq!(strict[0]["lost"], 1); + let wide = loss_by_outage(&ops, &topology, &sweep, RECOVERY_WINDOW_MS); + assert_eq!(wide[0]["written"], 3); + assert_eq!(wide[0]["lost"], 1); + } + + #[test] + fn runtime_is_the_name_prefix() { + assert_eq!(runtime("rust-2"), "rust"); + assert_eq!(runtime("go-0"), "go"); + } + + #[test] + fn grants_and_queries_are_not_writes() { + assert!(!is_write("grant")); + assert!(!is_write("query")); + assert!(is_write("update")); + } + + #[test] + fn payload_summary_splits_creates_from_updates() { + let ops = vec![ + json!({"kind":"create","ok":true,"payload_bytes":1000}), + json!({"kind":"create","ok":true,"payload_bytes":3000}), + json!({"kind":"update","ok":true,"payload_bytes":30}), + json!({"kind":"create","ok":false,"payload_bytes":9999}), + ]; + let got = payload_summary(&ops); + assert_eq!(got["create"]["n"], 2); + assert_eq!(got["create"]["sum"], 4000); + assert_eq!(got["update"]["n"], 1); + } + + #[test] + fn disk_fit_recovers_planted_terms() { + // growth = 500 bytes per write + 3x payload. Payload size shifts once + // partway through rather than alternating every write: alternating + // at a constant rate makes cumulative writes and cumulative payload + // asymptotically collinear as the sample count grows (the same + // pathology gate 2 exists to reject), so only a real regime shift + // keeps the two regressors separable here. + let mut ops = Vec::new(); + let mut du = Vec::new(); + let (mut w, mut pay) = (0u64, 0u64); + for i in 0..40u64 { + let bytes = if i < 20 { 1_000 } else { 9_000 }; + ops.push( + json!({"kind":"create","ok":true,"wall_ts_ms": i*1000, "payload_bytes": bytes}), + ); + w += 1; + pay += bytes; + du.push(json!({"node":"rust-0","wall_ts_ms": i*1000, "bytes": 500*w + 3*pay})); + } + let fit = &disk_fit(&ops, &du)[0]; + assert_eq!(fit["node"], "rust-0"); + assert!(fit["fitted"].as_bool().unwrap()); + assert!((fit["overhead_bytes_per_write"].as_f64().unwrap() - 500.0).abs() < 1.0); + assert!((fit["amplification_per_payload_byte"].as_f64().unwrap() - 3.0).abs() < 0.01); + } + + #[test] + fn disk_fit_is_suppressed_on_a_fixed_size_run() { + let mut ops = Vec::new(); + let mut du = Vec::new(); + let (mut w, mut pay) = (0u64, 0u64); + for i in 0..40u64 { + ops.push( + json!({"kind":"create","ok":true,"wall_ts_ms": i*1000, "payload_bytes": 1_200}), + ); + w += 1; + pay += 1_200; + du.push(json!({"node":"rust-0","wall_ts_ms": i*1000, "bytes": 500*w + 3*pay})); + } + let fit = &disk_fit(&ops, &du)[0]; + assert!( + !fit["fitted"].as_bool().unwrap(), + "a single payload size must not produce a fit" + ); + assert_eq!(fit["reason"], "payload sizes do not vary"); + } + + #[test] + fn disk_fit_requires_a_minimum_sample_count() { + let mut ops = Vec::new(); + let mut du = Vec::new(); + let (mut w, mut pay) = (0u64, 0u64); + for i in 0..5u64 { + let bytes = if i % 2 == 0 { 1_000 } else { 9_000 }; + ops.push( + json!({"kind":"create","ok":true,"wall_ts_ms": i*1000, "payload_bytes": bytes}), + ); + w += 1; + pay += bytes; + du.push(json!({"node":"rust-0","wall_ts_ms": i*1000, "bytes": 500*w + 3*pay})); + } + let fit = &disk_fit(&ops, &du)[0]; + assert!(!fit["fitted"].as_bool().unwrap()); + assert_eq!( + fit["reason"], + format!("fewer than {MIN_DISK_SAMPLES} disk samples") + ); + } + + #[test] + fn disk_fit_is_suppressed_when_regressors_are_collinear_at_scale() { + // Payload alternates between two close sizes (so the "sizes do not + // vary" gate does not fire), but cumulative writes and cumulative + // payload still move in near-lockstep over enough samples to clear + // the minimum-count gate: the raw determinant is enormous at this + // scale, but the two regressors remain unseparable. + let mut ops = Vec::new(); + let mut du = Vec::new(); + let (mut w, mut pay) = (0u64, 0u64); + for i in 0..20u64 { + let bytes = if i % 2 == 0 { 1_000_000 } else { 1_010_000 }; + ops.push( + json!({"kind":"create","ok":true,"wall_ts_ms": i*1000, "payload_bytes": bytes}), + ); + w += 1; + pay += bytes; + du.push(json!({"node":"rust-0","wall_ts_ms": i*1000, "bytes": 500*w + pay/1000})); + } + let fit = &disk_fit(&ops, &du)[0]; + assert!(!fit["fitted"].as_bool().unwrap()); + assert_eq!(fit["reason"], "regressors are collinear"); + } + + #[test] + fn disk_fit_rejects_a_borderline_tolerance_case() { + // Same planted terms as `disk_fit_recovers_planted_terms` (500/write + // + 3x payload), but the payload size shifts after 6 of 40 writes + // instead of 20. Computed tolerance ~= 0.0095 (still below + // MIN_TOLERANCE = 0.1, so this must stay suppressed). Nothing else + // in this file pins MIN_TOLERANCE's own value: dropping it from 0.1 + // to 1e-7 leaves every other test green but would let this + // borderline case through to a fit. + let mut ops = Vec::new(); + let mut du = Vec::new(); + let (mut w, mut pay) = (0u64, 0u64); + for i in 0..40u64 { + let bytes = if i < 6 { 1_000 } else { 9_000 }; + ops.push( + json!({"kind":"create","ok":true,"wall_ts_ms": i*1000, "payload_bytes": bytes}), + ); + w += 1; + pay += bytes; + du.push(json!({"node":"rust-0","wall_ts_ms": i*1000, "bytes": 500*w + 3*pay})); + } + let fit = &disk_fit(&ops, &du)[0]; + assert!(!fit["fitted"].as_bool().unwrap()); + assert_eq!(fit["reason"], "regressors are collinear"); + } + + #[test] + fn disk_fit_rejects_a_negative_fitted_term() { + // growth = 6000 bytes per write - 1x payload: impossible, but the + // payload size shifts partway through the run (as in the recovered- + // terms test) so the regressors are not collinear, and gate 2 lets + // this through; only the non-negativity backstop catches it. Disk + // usage does not just grow here: it rises to ~1.10M by i=19, then + // the payload shift outpaces the fixed 6000/write term and it falls + // to ~1.04M by i=39 -- exactly the shape the negative amplification + // predicts, not a real du series, which is the point of the test. + let mut ops = Vec::new(); + let mut du = Vec::new(); + let (mut w, mut pay) = (0u64, 0u64); + for i in 0..40u64 { + let bytes = if i < 20 { 1_000 } else { 9_000 }; + ops.push( + json!({"kind":"create","ok":true,"wall_ts_ms": i*1000, "payload_bytes": bytes}), + ); + w += 1; + pay += bytes; + du.push( + json!({"node":"rust-0","wall_ts_ms": i*1000, "bytes": 1_000_000 + 6000*w - pay}), + ); + } + let fit = &disk_fit(&ops, &du)[0]; + assert!(!fit["fitted"].as_bool().unwrap()); + assert_eq!(fit["reason"], "fitted term is negative"); + } + + #[test] + fn disk_fit_is_never_recorded_when_payload_bytes_is_always_zero() { + // Runs written before `payload_bytes` existed parse every op's + // missing field as 0, which must not be reported as "sizes do not + // vary" (a fixed-size profile): the honest reason is that payload + // was never recorded at all. + let mut ops = Vec::new(); + let mut du = Vec::new(); + for i in 0..20u64 { + ops.push(json!({"kind":"create","ok":true,"wall_ts_ms": i*1000})); + du.push(json!({"node":"rust-0","wall_ts_ms": i*1000, "bytes": 1_000 * i})); + } + let fit = &disk_fit(&ops, &du)[0]; + assert!(!fit["fitted"].as_bool().unwrap()); + assert_eq!(fit["reason"], "payload was never recorded"); + } + + #[test] + fn disk_fit_is_independent_per_node() { + // Same mesh-wide writes; rust-0's disk grows with the planted terms + // from `disk_fit_recovers_planted_terms`, go-0's with the impossible + // series from `disk_fit_rejects_a_negative_fitted_term`. A mixed + // mesh's nodes need not agree, so each must get its own verdict. + let mut ops = Vec::new(); + let mut du = Vec::new(); + let (mut w, mut pay) = (0u64, 0u64); + for i in 0..40u64 { + let bytes = if i < 20 { 1_000 } else { 9_000 }; + ops.push( + json!({"kind":"create","ok":true,"wall_ts_ms": i*1000, "payload_bytes": bytes}), + ); + w += 1; + pay += bytes; + du.push(json!({"node":"rust-0","wall_ts_ms": i*1000, "bytes": 500*w + 3*pay})); + du.push(json!({"node":"go-0","wall_ts_ms": i*1000, "bytes": 1_000_000 + 6000*w - pay})); + } + let fit = disk_fit(&ops, &du); + let by_node: HashMap<&str, &Value> = fit + .as_array() + .unwrap() + .iter() + .map(|f| (f["node"].as_str().unwrap(), f)) + .collect(); + assert!(by_node["rust-0"]["fitted"].as_bool().unwrap()); + assert!(!by_node["go-0"]["fitted"].as_bool().unwrap()); + assert_eq!(by_node["go-0"]["reason"], "fitted term is negative"); + } +} diff --git a/crates/soak/src/tags.rs b/crates/soak/src/tags.rs new file mode 100644 index 0000000..4dd747e --- /dev/null +++ b/crates/soak/src/tags.rs @@ -0,0 +1,106 @@ +//! Known-cause tags for divergent docs, decided live from the doc's last +//! successful write and the mesh's outage windows. +//! +//! `write-during-outage`: a member of the divergent pair other than the +//! writer was down when the doc was last written. `write-during-recovery`: +//! the writer or a pair member had come back less than [`RECOVERY_MS`] +//! before the write (a node answers queries before its replicator link is +//! up). Both are the M0 finding; anything untagged is the alarm. +use serde::Serialize; + +/// Writes this soon after a recovery count as made during it. +pub const RECOVERY_MS: u64 = 30_000; + +/// One node outage: `up` is `None` while the node is still down. +#[derive(Clone, Debug)] +pub struct Outage { + pub node: usize, + pub down_ms: u64, + pub up_ms: Option, +} + +#[derive(Clone, Copy, Debug, PartialEq, Eq, Serialize)] +#[serde(rename_all = "kebab-case")] +pub enum Tag { + WriteDuringOutage, + WriteDuringRecovery, +} + +/// Tag for a doc whose last write ran on `writer` at `wall_ms`, diverging on +/// the pair `(a, b)`. +pub fn tag(writer: usize, wall_ms: u64, pair: (usize, usize), outages: &[Outage]) -> Option { + let involved = |n: usize| n == writer || n == pair.0 || n == pair.1; + let peer_down = outages.iter().any(|o| { + o.node != writer + && involved(o.node) + && o.down_ms <= wall_ms + && o.up_ms.is_none_or(|up| wall_ms <= up) + }); + if peer_down { + return Some(Tag::WriteDuringOutage); + } + let recovering = outages.iter().any(|o| { + involved(o.node) + && o.up_ms + .is_some_and(|up| up <= wall_ms && wall_ms - up <= RECOVERY_MS) + }); + recovering.then_some(Tag::WriteDuringRecovery) +} + +#[cfg(test)] +mod tests { + use super::*; + + const R0: usize = 0; + const R1: usize = 1; + const G0: usize = 2; + + fn outage(node: usize, down: u64, up: Option) -> Outage { + Outage { + node, + down_ms: down, + up_ms: up, + } + } + + #[test] + fn peer_down_at_write_time() { + let o = [outage(G0, 1_000, Some(20_000))]; + assert_eq!(tag(R0, 5_000, (R0, G0), &o), Some(Tag::WriteDuringOutage)); + // Still down (no up yet) counts too. + let o = [outage(G0, 1_000, None)]; + assert_eq!(tag(R0, 5_000, (R0, G0), &o), Some(Tag::WriteDuringOutage)); + } + + #[test] + fn peer_or_writer_recovered_just_before() { + let o = [outage(G0, 1_000, Some(20_000))]; + assert_eq!( + tag(R0, 25_000, (R0, G0), &o), + Some(Tag::WriteDuringRecovery) + ); + // Writer's own recovery, pair member elsewhere. + let o = [outage(R0, 1_000, Some(20_000))]; + assert_eq!( + tag(R0, 40_000, (R0, R1), &o), + Some(Tag::WriteDuringRecovery) + ); + // Past the window: nothing. + assert_eq!(tag(R0, 60_000, (R0, R1), &o), None); + } + + #[test] + fn outages_of_uninvolved_nodes_do_not_count() { + let o = [outage(R1, 1_000, Some(20_000))]; + assert_eq!(tag(R0, 5_000, (R0, G0), &o), None); + assert_eq!(tag(R0, 25_000, (R0, G0), &o), None); + } + + #[test] + fn outage_wins_over_recovery() { + let o = [outage(G0, 1_000, Some(20_000)), outage(R0, 21_000, None)]; + // Writer R0 cannot really write while down, but if it did the peer's + // recovery is the weaker claim. + assert_eq!(tag(G0, 25_000, (R0, G0), &o), Some(Tag::WriteDuringOutage)); + } +}