diff --git a/benchmark-results/2026-10-07-g4-dualpath/README.md b/benchmark-results/2026-10-07-g4-dualpath/README.md new file mode 100644 index 00000000..8a068b23 --- /dev/null +++ b/benchmark-results/2026-10-07-g4-dualpath/README.md @@ -0,0 +1,61 @@ +# G4 passes: 6.184x, once the FFN gate and up projections reach the device + +**Measured 2026-10-07 on an RTX 4090 (cc 8.9, driver 12080), Runpod.** Models revision +`c107cd14f9c1ee8f7a446aa861eddbb77b42959e`. Same model, same prompts, same host class and the same +`--arm both` protocol as `benchmark-results/2026-10-07-g4-dispatch`, which is the baseline here. + +The PTX module is **byte-identical** across the two runs -- kernel sha256 +`4c534e622452df95ca7fbb3accc65246be54e2c74d53714d98a03be6d31e8e0e` in both. The change was Java +side only, so nothing in the device code moved and the comparison isolates the routing. + +## Before and after + +| | baseline | dual path routed | delta | +| --- | --- | --- | --- | +| **accelerated decode** | 9.02 tok/s | **31.26 tok/s** | **+246.6%** | +| control decode | 5.05 tok/s | 5.06 tok/s | +0.2% | +| **G4** | **1.788x — FAILED** | **6.184x — PASSED** | gate is 3.00x | +| `qualified` | false | **true** | +| launches / decode step | 241.0 | 321.0 | +33.2% | +| transfers / decode step | 402.0 | 522.0 | +29.9% | +| activation bytes / decode step | 12,660,712 | 15,398,952 | +21.6% | +| Q4_K decode projections per layer | 4.05 | **6.07** | predicted 6.00 | + +The control arm moving 0.2% is what makes the rest readable: the denominator is stable, so the +accelerated arm's 3.47x is not an artefact of a quieter host. + +## What it shows + +**The dispatch cost rose and it did not matter.** 33% more launches and 30% more transfers bought a +3.47x faster accelerated arm, because the two projections that moved are 53.3% of a layer's +projection arithmetic -- `ffn_gate` and `ffn_up`, 8192x2560 each on Granite 4.1 3B. The earlier +reading of 1.788x as a dispatch ceiling was wrong: dispatch was never the binding constraint at that +point, an unimplemented code path was. + +`Q4_K/DECODE_PROJECTION` per layer landing on **6.07** against the 6.00 predicted from the GGUF +header is the direct confirmation that the dual path is now taken, independent of any timing. + +## Correctness is not the claim + +The previous behaviour was correct. `TensorOps.ggufDualMatmul` computed the gate and up projections +on the CPU and produced right answers; it was slower, not wrong. So this change has no correctness +credit to claim and would have earned nothing had the speed not moved -- it is kept because +31.26 tok/s beats 9.02 tok/s, and for no other reason. + +## What is still open + +**Device-resident activations are no longer the next lever, and the number that says so is here.** +522 transfers per decode step moving 15.4 MB remains the standing overhead, but it now sits on top +of an arm that passes its gate at 2.06x the threshold. Collapsing those transfers to about 2 is +still arithmetic rather than a measurement, and it needs the elementwise kernels plus a seam that +hands off a layer; it should be prioritised against G1 rather than assumed next. + +**G1 is untouched.** This run reports `tokenParity: null` -- it is a decode-speed measurement and +says nothing about the attention argmax. The path passes G4 and still cannot ship until G1 is +settled. + +**The declined-projection counter did not reach this report.** `CudaRoutingCounters.declined` was +added and incremented, but `CudaKernelGateCli.Routing` never serialised it, so this run printed +`totalDeclinedProjections = None` -- the counter built to make a silent fallback visible was itself +invisible in the artifact. Fixed in the same change as this record, and verified off-device: the +report now carries `declinedProjections` and `totalDeclinedProjections`. diff --git a/benchmark-results/2026-10-07-g4-dualpath/rust-ptx-capability.json b/benchmark-results/2026-10-07-g4-dualpath/rust-ptx-capability.json new file mode 100644 index 00000000..d0a9e5b5 --- /dev/null +++ b/benchmark-results/2026-10-07-g4-dualpath/rust-ptx-capability.json @@ -0,0 +1,74 @@ +{ + "schemaVersion" : 2, + "createdAt" : "2026-10-07T17:25:56.514564446Z", + "policyVersion" : "gpu-large-model-rust-ptx-v2", + "modelsRevision" : "c107cd14f9c1ee8f7a446aa861eddbb77b42959e", + "mode" : "capability", + "accelerated" : true, + "refusalReason" : "", + "device" : { + "name" : "NVIDIA GeForce RTX 4090", + "computeCapability" : 89, + "globalMemoryBytes" : 25261047808, + "driverVersion" : 12080 + }, + "kernel" : { + "sha256" : "4c534e622452df95ca7fbb3accc65246be54e2c74d53714d98a03be6d31e8e0e", + "ptxTarget" : "sm_80", + "toolchain" : "nightly-2026-09-17", + "entryPoints" : [ "models_q4k_decode_projection", "models_q6k_decode_projection", "models_gqa_decode_attention" ] + }, + "environment" : { + "host" : "817bdc3ee892", + "osName" : "Linux", + "osVersion" : "6.8.0-117-generic", + "architecture" : "amd64", + "cpuModel" : "AMD EPYC 7K62 48-Core Processor", + "processors" : 11, + "physicalMemoryBytes" : 30999998464, + "maxHeapBytes" : 7751073792, + "javaVersion" : "25.0.4.1", + "javaVendor" : "Eclipse Adoptium", + "vmName" : "OpenJDK 64-Bit Server VM" + }, + "configuration" : null, + "routing" : { + "acceleratedOperations" : { }, + "refusals" : { }, + "totalAcceleratedOperations" : 0, + "inert" : true, + "kernelLaunches" : 0, + "hostToDeviceTransfers" : 0, + "deviceToHostTransfers" : 0, + "hostToDeviceBytes" : 0, + "deviceToHostBytes" : 0, + "weightUploads" : 0, + "weightUploadBytes" : 0, + "decodeSteps" : 0, + "decodeProjections" : 0, + "measuredDecodeSteps" : 0, + "decodeStepsMarked" : false, + "launchesPerDecodeStep" : 0.0, + "transfersPerDecodeStep" : 0.0, + "activationBytesPerDecodeStep" : 0.0 + }, + "readinessMillis" : 523, + "parity" : null, + "decode" : null, + "memory" : null, + "gates" : { + "routingObservability" : { + "gate" : "G2", + "passed" : true, + "evidence" : "no accelerated operations recorded" + }, + "startupHonesty" : { + "gate" : "G5", + "passed" : true, + "evidence" : "523 ms readiness against a 120000 ms ceiling" + }, + "tokenParity" : null, + "decodeSpeedup" : null + }, + "qualified" : true +} diff --git a/benchmark-results/2026-10-07-g4-dualpath/rust-ptx-decode.json b/benchmark-results/2026-10-07-g4-dualpath/rust-ptx-decode.json new file mode 100644 index 00000000..39f1d7f7 --- /dev/null +++ b/benchmark-results/2026-10-07-g4-dualpath/rust-ptx-decode.json @@ -0,0 +1,148 @@ +{ + "schemaVersion" : 2, + "createdAt" : "2026-10-07T17:33:23.492141108Z", + "policyVersion" : "gpu-large-model-rust-ptx-v2", + "modelsRevision" : "c107cd14f9c1ee8f7a446aa861eddbb77b42959e", + "mode" : "decode", + "accelerated" : true, + "refusalReason" : "", + "device" : { + "name" : "NVIDIA GeForce RTX 4090", + "computeCapability" : 89, + "globalMemoryBytes" : 25261047808, + "driverVersion" : 12080 + }, + "kernel" : { + "sha256" : "4c534e622452df95ca7fbb3accc65246be54e2c74d53714d98a03be6d31e8e0e", + "ptxTarget" : "sm_80", + "toolchain" : "nightly-2026-09-17", + "entryPoints" : [ "models_q4k_decode_projection", "models_q6k_decode_projection", "models_gqa_decode_attention" ] + }, + "environment" : { + "host" : "817bdc3ee892", + "osName" : "Linux", + "osVersion" : "6.8.0-117-generic", + "architecture" : "amd64", + "cpuModel" : "AMD EPYC 7K62 48-Core Processor", + "processors" : 11, + "physicalMemoryBytes" : 30999998464, + "maxHeapBytes" : 7751073792, + "javaVersion" : "25.0.4.1", + "javaVendor" : "Eclipse Adoptium", + "vmName" : "OpenJDK 64-Bit Server VM" + }, + "configuration" : { + "mode" : "decode", + "modelPath" : "/work/model.gguf", + "modelSha256" : "662b0626cd58f443baea23559b469df6576a81d349649c59413b36a9fb32eb29", + "modelBytes" : 2099501664, + "promptsPath" : "/work/models/benchmark-results/2026-09-18-gpu-large-model/prompts.txt", + "promptsSha256" : "d0c0267f1eeb96b97a6e26f964d4e06c0e967ad7567e40881e54c791d74c7016", + "promptCount" : 20, + "maxTokens" : 64, + "warmupTokens" : 16, + "contextLength" : 4096, + "arm" : "both", + "sampling" : "greedy-argmax-lowest-index-v1", + "cudaDisabledProperty" : false, + "cudaDeviceOrdinal" : 0, + "jvmArguments" : [ "--add-modules=jdk.incubator.vector", "--enable-native-access=ALL-UNNAMED", "-XX:NativeMemoryTracking=summary" ], + "workingDirectory" : "/work/models" + }, + "routing" : { + "acceleratedOperations" : { + "Q4_K/DECODE_PROJECTION" : 306000, + "Q6_K/PREFILL_PROJECTION" : 1040, + "Q4_K/PREFILL_PROJECTION" : 6240, + "Q6_K/DECODE_PROJECTION" : 52296, + "F32/DECODE_ATTENTION" : 2040000 + }, + "refusals" : { }, + "totalAcceleratedOperations" : 2405576, + "inert" : false, + "kernelLaunches" : 416576, + "hostToDeviceTransfers" : 260456, + "deviceToHostTransfers" : 416576, + "hostToDeviceBytes" : 13756064960, + "deviceToHostBytes" : 8247738368, + "weightUploads" : 281, + "weightUploadBytes" : 2095104000, + "decodeSteps" : 1260, + "decodeProjections" : 358296, + "measuredDecodeSteps" : 1260, + "decodeStepsMarked" : true, + "launchesPerDecodeStep" : 321.0, + "transfersPerDecodeStep" : 522.0, + "activationBytesPerDecodeStep" : 1.5398952E7 + }, + "readinessMillis" : 353, + "parity" : null, + "decode" : { + "accelerated" : { + "arm" : "accelerated", + "accelerated" : true, + "promptCount" : 20, + "promptTokens" : 512, + "generatedTokens" : 1280, + "decodeSteps" : 1260, + "sequencesHittingEndOfGeneration" : 1, + "prefillNanos" : 11757362556, + "decodeNanos" : 40304758701, + "wallClockNanos" : 54559411329, + "prefillTokensPerSecond" : 43.547181399004906, + "decodeTokensPerSecond" : 31.26181722975402, + "residentBytesAtEnd" : 530350080, + "heapUsedBytesAtEnd" : 370705272, + "peakDeviceBytes" : 2097012576, + "deviceTotalMemoryBytes" : 25261047808 + }, + "control" : { + "arm" : "control", + "accelerated" : false, + "promptCount" : 20, + "promptTokens" : 512, + "generatedTokens" : 1280, + "decodeSteps" : 1260, + "sequencesHittingEndOfGeneration" : 1, + "prefillNanos" : 68485761145, + "decodeNanos" : 249248379237, + "wallClockNanos" : 324196877538, + "prefillTokensPerSecond" : 7.4760065660361, + "decodeTokensPerSecond" : 5.055198368218547, + "residentBytesAtEnd" : 640892928, + "heapUsedBytesAtEnd" : 241280936, + "peakDeviceBytes" : 0, + "deviceTotalMemoryBytes" : 25261047808 + }, + "speedup" : 6.184093076602786, + "threshold" : 3.0, + "wallClockMillis" : 378763 + }, + "memory" : { + "peakProcessRssBytes" : 2710339584, + "residentBytes" : 640897024, + "heapUsedBytes" : 241280936, + "peakDeviceBytes" : 2097012576, + "deviceTotalMemoryBytes" : 25261047808, + "reading" : "peakDeviceBytes covers kernel-owned allocations only (weights, staging scratch, per-call attention buffers); the CUDA context and loaded module are not included, so it is a lower bound. peakProcessRssBytes is process-wide and therefore not separable per arm in a two-arm run: use --arm accelerated and --arm control in separate processes for per-arm host peaks." + }, + "gates" : { + "routingObservability" : { + "gate" : "G2", + "passed" : true, + "evidence" : "2405576 accelerated operations" + }, + "startupHonesty" : { + "gate" : "G5", + "passed" : true, + "evidence" : "353 ms readiness against a 120000 ms ceiling" + }, + "tokenParity" : null, + "decodeSpeedup" : { + "gate" : "G4", + "passed" : true, + "evidence" : "6.184x decode against a 3.00x threshold" + } + }, + "qualified" : true +} diff --git a/benchmark-results/2026-10-07-g4-dualpath/worker.log b/benchmark-results/2026-10-07-g4-dualpath/worker.log new file mode 100644 index 00000000..90b4c529 --- /dev/null +++ b/benchmark-results/2026-10-07-g4-dualpath/worker.log @@ -0,0 +1,79 @@ +[17:19:42] nvidia-smi: +NVIDIA GeForce RTX 4090, 8.9, 570.211.01 +[17:19:42] installing jdk 25 +[17:23:41] jdk: openjdk version "25.0.4.1" 2026-08-18 LTS +[17:23:41] installing pinned rust nightly-2026-09-17 +[17:24:00] rust: rustc 1.100.0-nightly (923c95cdf 2026-09-16) +[17:24:00] fetching source +[17:24:02] source at /work/models, revision c107cd14f9c1ee8f7a446aa861eddbb77b42959e +[17:24:02] === STEP 1: ptxas assembly (cheap gate) === + Downloaded rustc-literal-escaper v0.0.8 + Downloaded rustc-demangle v0.1.28 + Downloaded hashbrown v0.17.1 + Downloaded libc v0.2.189 + Compiling compiler_builtins v0.1.160 (/root/.rustup/toolchains/nightly-2026-09-17-x86_64-unknown-linux-gnu/lib/rustlib/src/rust/library/compiler-builtins/compiler-builtins) + Compiling core v0.0.0 (/root/.rustup/toolchains/nightly-2026-09-17-x86_64-unknown-linux-gnu/lib/rustlib/src/rust/library/core) + Compiling models-cuda-kernels v0.3.46 (/work/models/backend-cuda/src/main/rust/models-cuda-kernels) + Finished `release` profile [optimized] target(s) in 20.47s +[17:25:55] ptx: backend-cuda/build/generated/cuda-resources/META-INF/models/cuda/models-cuda-kernels.ptx (64104 bytes) +[17:25:55] ptxas OK +[17:25:55] === STEP 2: capability gate === +WARNING: Using incubator modules: jdk.incubator.vector +PASS cuda-kernel-gate mode=capability device=NVIDIA GeForce RTX 4090 cc=89 kernel=4c534e622452 readiness=523 ms report=/work/capability.json +[17:25:57] capability report uploaded +[17:25:57] === STEP 2a: Q6_K device parity (gate before any model) === +CudaQ6KDeviceParityTest > a Q6_K row projection is bit-exact against the CPU control at every width > 1 super-block(s) PASSED + +CudaQ6KDeviceParityTest > a Q6_K row projection is bit-exact against the CPU control at every width > 2 super-block(s) PASSED + +CudaQ6KDeviceParityTest > a Q6_K row projection is bit-exact against the CPU control at every width > 3 super-block(s) PASSED + +CudaQ6KDeviceParityTest > a Q6_K row projection is bit-exact against the CPU control at every width > 32 super-block(s) PASSED + +CudaQ6KDeviceParityTest > a Q6_K row projection is bit-exact against the CPU control at every width > 33 super-block(s) PASSED + +CudaQ6KDeviceParityTest > a Q6_K row projection is bit-exact against the CPU control at every width > 48 super-block(s) PASSED + +BUILD SUCCESSFUL in 9s +13 actionable tasks: 3 executed, 10 up-to-date +Consider enabling configuration cache to speed up this build: https://docs.gradle.org/9.4.1/userguide/configuration_cache_enabling.html +[17:26:06] Q6_K device parity OK +[17:26:06] === fetching model (granite 4.1 3b q4_k_m, 1.96GB) === +[17:27:02] model sha verified +[17:27:02] === STEP 4: decode gate G4, --arm both === +WARNING: Using incubator modules: jdk.incubator.vector +Oct 07, 2026 5:27:04 PM com.integrallis.vectors.core.PanamaVectorUtilSupportProvider create +INFO: vectors-core: Using Panama Vector API SIMD provider (vector bits: 256) +Oct 07, 2026 5:27:04 PM com.integrallis.vectors.core.VectorizationProvider logActiveToggles +INFO: vectors-core: provider=PanamaVectorUtilSupport panama=true maxBits=256 preferredBits=256 fastVectorFMA=true fastScalarFMA=true sve=false ggufParallel=true ggufParallelThreshold=1048576 ggufExecutor=persistent ggufThreads=11 ggufChunksPerThread=2 mappedKQuantLongOffsets=auto(q4=true,q5=false,q6=false) q4ShortPairwiseSupported=true q4UnsignedPairwiseSupported=true toggles=[(defaults ? no -Dvectors.* overrides)] +PASS cuda-kernel-gate mode=decode device=NVIDIA GeForce RTX 4090 arm=both accelerated=31.26 tok/s control=5.06 tok/s speedup=6.184 threshold=3.00 report=/work/decode.json + routing: launches/step=321.000 transfers/step=522.000 activationBytes/step=15398952.000 (1260 measured steps, 358296 decode projections, 281 weight uploads) +[17:33:24] decode report uploaded +[17:33:24] decode gate rc=0 +[17:33:24] === the numbers this run exists for === + --- routing counters, the numbers this run exists for --- + routing.kernelLaunches = 416576 + routing.hostToDeviceTransfers = 260456 + routing.deviceToHostTransfers = 416576 + routing.hostToDeviceBytes = 13756064960 + routing.deviceToHostBytes = 8247738368 + routing.weightUploadBytes = 2095104000 + routing.decodeSteps = 1260 + routing.measuredDecodeSteps = 1260 + routing.decodeStepsMarked = True + routing.launchesPerDecodeStep = 321.0 + routing.transfersPerDecodeStep = 522.0 + routing.activationBytesPerDecodeStep = 15398952.0 + --- declined projections (new: the counter that would have caught this) --- + totalDeclinedProjections = None + (none reported) + --- arms --- + decode.accelerated: prefill=43.547181399004906 decode=31.26181722975402 tok/s steps=1260 peakDevice=2097012576 + decode.control: prefill=7.4760065660361 decode=5.055198368218547 tok/s steps=1260 peakDevice=0 + --- verdict --- + accelerated = True + refusalReason = + gates = {"routingObservability": {"gate": "G2", "passed": true, "evidence": "2405576 accelerated operations"}, "startupHonesty": {"gate": "G5", "passed": true, "evidence": "353 ms readiness against a 120000 ms ceiling"}, "tokenParity": null, "decodeSpeedup": {"gate": "G4", "passed": true, "evidence": "6.184x decode against a 3.00x threshold"}} + qualified = True + device = {"name": "NVIDIA GeForce RTX 4090", "computeCapability": 89, "globalMemoryBytes": 25261047808, "driverVersion": 12080} +[17:33:24] FINAL_STATUS=COMPLETE diff --git a/models-bench/src/main/java/com/integrallis/models/bench/CudaKernelGateCli.java b/models-bench/src/main/java/com/integrallis/models/bench/CudaKernelGateCli.java index d39328b4..2088d6a4 100644 --- a/models-bench/src/main/java/com/integrallis/models/bench/CudaKernelGateCli.java +++ b/models-bench/src/main/java/com/integrallis/models/bench/CudaKernelGateCli.java @@ -576,7 +576,9 @@ private static Routing routing( return new Routing( counters.acceleratedOperations(), counters.refusals(), + counters.declinedProjections(), counters.totalAcceleratedOperations(), + counters.totalDeclinedProjections(), counters.inert(), counters.kernelLaunches(), counters.hostToDeviceTransfers(), @@ -598,6 +600,8 @@ private static Routing emptyRouting(CudaRoutingRecorder.Measured measured) { return new Routing( Map.of(), Map.of(), + Map.of(), + 0L, 0L, true, 0L, @@ -829,7 +833,9 @@ record RunConfiguration( record Routing( Map acceleratedOperations, Map refusals, + Map declinedProjections, long totalAcceleratedOperations, + long totalDeclinedProjections, boolean inert, long kernelLaunches, long hostToDeviceTransfers, @@ -849,6 +855,7 @@ record Routing( Routing { acceleratedOperations = Map.copyOf(acceleratedOperations); refusals = Map.copyOf(refusals); + declinedProjections = Map.copyOf(declinedProjections); } }