diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index d198c3c..9c3b760 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -57,3 +57,20 @@ jobs: - name: Check dependency hygiene run: bun run check:deps + + rust-validation: + name: Cross-validate vectors against the Rust reference + runs-on: ubuntu-latest + + steps: + - name: Checkout + uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7 + + # The runner image ships a stable Rust toolchain + - name: Validate against dcbor (default build) + working-directory: tests/rust-validation + run: cargo run --release --locked -- ../vectors + + - name: Validate against dcbor (num-bigint build) + working-directory: tests/rust-validation + run: cargo run --release --locked --features bignum -- ../vectors diff --git a/.gitignore b/.gitignore index 4d59281..4660bc6 100644 --- a/.gitignore +++ b/.gitignore @@ -8,7 +8,7 @@ node_modules # Build artifacts dist coverage -# Generated TypeDoc output; retain handwritten architecture notes. +# Generated TypeDoc output /docs/* *.tsbuildinfo diff --git a/.size-limit.json b/.size-limit.json index 5dfdcf2..e1e96b5 100644 --- a/.size-limit.json +++ b/.size-limit.json @@ -2,12 +2,12 @@ { "name": "ESM entry (import *), minified + gzipped", "path": "dist/index.mjs", - "limit": "10 kB" + "limit": "11.5 kB" }, { "name": "decode-only ({ decodeCbor }), minified + gzipped", "path": "dist/index.mjs", "import": "{ decodeCbor }", - "limit": "6.5 kB" + "limit": "8 kB" } ] \ No newline at end of file diff --git a/CHANGELOG.md b/CHANGELOG.md index 1829178..466ebf1 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -1,5 +1,80 @@ # Changelog +## Unreleased + +### Changed + +- **Decoding keeps a leading U+FEFF** in a text string, as the reference's + `String::from_utf8` does; the WHATWG `TextDecoder` default stripped it, so + `64efbbbf61` did not round-trip. +- **`cbor(string)` keeps the string as given; encoding normalizes to NFC** + (the reference's `cbor_data` does the same). `expectText`, `cborEquals`, + `diagnostic` and `hexAnnotated` now see the original string; the wire + bytes are unchanged. Byte-string annotations in `hexAnnotated` keep astral + characters (`"😀"`) instead of dotting them. +- **`InvalidUtf8` messages mirror `core::str::Utf8Error`** + (`invalid utf-8 sequence of 1 bytes from index 3`, `incomplete utf-8 byte + sequence from index 0`) instead of the host decoder's text. +- **Float heads are validated with the reference's `validate_canonical_*` + predicates** and decode to the node its `From`/`From` build. A + whole-valued f32 head at or beyond 2^31 (or an f64 head at or beyond + 2^63) is accepted and reduces to an integer where one fits: `fa4f000001` + decodes to `2147483904` and re-encodes as `1a80000100`; `fa4f800000` + stays a float. Before, every such head was `NonCanonicalNumeric`. +- **Float diagnostics round exact decimal ties up**, as Rust's `{:?}` does + (`f9000a` prints `5.960464477539063e-7`, not `…062e-7`); `simpleName` + prints `inf`/`-inf`/`42.0` like `Simple`'s `Debug`. +- **`CborDate` holds whole seconds plus nanoseconds**, the reference's + `chrono::DateTime` model. A parsed leap second displays as `:60` + (`2023-12-25T10:30:60Z`), `equals`/`compare` distinguish it from the next + second, a sub-second fraction can no longer move the displayed second, and + the range check applies to the truncated whole seconds (`MIN - 0.5` is + `MIN`). `fromUntaggedCbor`/`fromTaggedCbor` return a new instance. +- **`CborDate.fromEpochSeconds(NaN)` and a decoded tag-1 `NaN` are the + epoch** (`c100`, `1970-01-01`), as the reference's saturating cast makes + them; ±Infinity is still `InvalidDate` (the reference panics). +- **`WrongTag` errors name both tags as the reference does.** `CborDate`'s + expected tag carries the global store's name for tag 1 (`date` once + `registerStandardTags()` has run, else `1`); `taggedValue(tag, …)` keeps a + named `Tag`'s name on the node (`CborTaggedType.tagName`, never on the + wire) so the actual tag is named too; `expectTaggedContent` accepts a + `Tag` and keeps its name. +- **`cborEquals` is structural** (`PartialEq for CBOR`): a whole-valued + float node is not equal to the integer it encodes as, a decomposed string + is not equal to its composed form, `NaN` equals `NaN`, tags compare by + value, maps entry by entry. It compared encodings before. +- **`registerStandardTags` registers unconditionally** through + `registerAll`, as `insert_all` does: it moves each standard name back to + its standard value instead of skipping a store that already had the name. + `registerAll` accepts any `Iterable`. +- **Tags are frozen values.** `Tag.from` returns a frozen object and a + `TagsStore` keeps frozen tags by identity (an unfrozen literal is copied). +- **One global tags store per process.** `getGlobalTagsStore()` keeps the + store on `globalThis` under `Symbol.for("@blockchaincommons/dcbor/global-tags-store@1")`, + so the ESM and CommonJS builds share it, as the reference's `GLOBAL_TAGS`. + +### Added + +- `expectUnsigned(cbor, { width, wrapNegative })` extracts into a fixed + width like `u8`…`u64::try_from`, including the reference's wrap of a + negative integer (`-1` → `255` at width 8; RUST_DIVERGENCES.md §1.1). + Without options the behaviour is unchanged. +- `TagsStore.clone()` mirrors `#[derive(Clone)]`: an independent copy that + shares frozen tags and summarizer functions. +- Harness: `format-vectors.json`, `date-vectors.json` and + `uint-vectors.json`; decode rejections pin the reference's error message; + a `bignum` feature runs the reference's `num-bigint` build; both builds + run in CI. Guard tests for the engine's Unicode version and a 1,000-deep + array. + +### Removed + +- The `TAG_*` constants with no counterpart in the reference (`TAG_UUID`, + `TAG_ENCODED_CBOR`, `TAG_SET`, `TAG_SELF_DESCRIBE_CBOR`, …; the reference + defines only `TAG_DATE`, `TAG_POSITIVE_BIGNUM` and `TAG_NEGATIVE_BIGNUM`). + `TAG_EPOCH_DATE_TIME` was a duplicate of `TAG_DATE`. `TAG_BINARY_UUID` + named IANA tag 257, which is the binary MIME message tag. + ## 1.0.0-beta.2 - 2026-09-13 The review against `dcbor` 0.25.2 aligned diagnostic formatting, date @@ -43,9 +118,9 @@ validation, and optional bignum tag registration. stored timestamp is the reference's `timestamp()` arithmetic (whole seconds plus nanoseconds over 10⁹), so a fraction encodes to the same bytes on both sides; `fromDate` computes a JS `Date`'s milliseconds the - same way. 57 `datestr/*` vectors run every form through the reference. + same way. 62 `datestr/*` vectors run every form through the reference. - **`CborDate.fromYmd` / `fromYmdHms` validate their components** - year - within ±262143, a calendar month and day, a time within 23:59:59 - and + within −262143…+262142, a calendar month and day, a time within 23:59:59 - and throw `InvalidDate` where the reference's `with_ymd_and_hms(…).unwrap()` panics. They used to roll over (`2023-13-01` became 2024-01-01) and mapped years 0–99 to 1900–1999. diff --git a/MIGRATION.md b/MIGRATION.md index 6857bcb..72cbefd 100644 --- a/MIGRATION.md +++ b/MIGRATION.md @@ -2,27 +2,27 @@ `@blockchaincommons/dcbor` is the redesigned successor to `@bcts/dcbor`. -## TL;DR checklist - -- [ ] Replace the dependency `@bcts/dcbor` with `@blockchaincommons/dcbor`; update imports. -- [ ] Import `diagnostic`/`hexAnnotated` from `@blockchaincommons/dcbor/diagnostic` and - `walk` & friends from `@blockchaincommons/dcbor/walk` (they left the root). -- [ ] `cborData(v)` → `encodeCbor(v)`; `toTaggedValue(t, v)` → `taggedValue(t, v)`. -- [ ] Replace `Cbor` *value* usages (`Cbor.from`, `Cbor.tryFromData`, - `Cbor.True`…) - the type survives, the namespace value is gone. -- [ ] Move instance-method calls to free functions (`c.isMap()` → `isMap(c)`, - `c.toText()` → `expectText(c)`, …) - the value keeps only - `toData()`/`toHex()`/`toString()`. -- [ ] Errors: `errorToString(e)`/`errorMsg(e)` → `e.message`; - `e.errorType.type` → `e.code`; `e.errorType.` → `e.details.`; - `new CborError({ type })` → `CborError.()`. -- [ ] Containers: `CborMap.new()/insert/containsKey/len` → - `new CborMap()/set/has/size`; `CborSet.insert/contains/fromArray` → - `add/has/from`; note `CborMap.get` now returns the stored `Cbor` node. -- [ ] Fix the two shapes that now THROW: plain `{tag, value}` object - literals and `taggedCbor()`-only objects (see below). -- [ ] *(optional)* adopt `tryDecode()` (non-throwing) and - `decodeWith(bytes, codec)` (typed decode, `@beta`). +## Summary + +- Replace the dependency `@bcts/dcbor` with `@blockchaincommons/dcbor`; update imports. +- Import `diagnostic`/`hexAnnotated` from `@blockchaincommons/dcbor/diagnostic` and + `walk` & friends from `@blockchaincommons/dcbor/walk` (they left the root). +- `cborData(v)` → `encodeCbor(v)`; `toTaggedValue(t, v)` → `taggedValue(t, v)`. +- Replace `Cbor` *value* usages (`Cbor.from`, `Cbor.tryFromData`, + `Cbor.True`…) - the type survives, the namespace value is gone. +- Move instance-method calls to free functions (`c.isMap()` → `isMap(c)`, + `c.toText()` → `expectText(c)`, …) - the value keeps only + `toData()`/`toHex()`/`toString()`. +- Errors: `errorToString(e)`/`errorMsg(e)` → `e.message`; + `e.errorType.type` → `e.code`; `e.errorType.` → `e.details.`; + `new CborError({ type })` → `CborError.()`. +- Containers: `CborMap.new()/insert/containsKey/len` → + `new CborMap()/set/has/size`; `CborSet.insert/contains/fromArray` → + `add/has/from`; `CborMap.get` returns the stored `Cbor` node. +- Fix the two shapes that throw: plain `{tag, value}` object literals and + `taggedCbor()`-only objects (see below). +- Optionally adopt `tryDecode()` (non-throwing) and + `decodeWith(bytes, codec)` (typed decode, `@beta`). --- @@ -41,7 +41,7 @@ and their options types), `@blockchaincommons/dcbor/walk` (`walk`, `Visitor`, `W `@blockchaincommons/dcbor/debug` (`installDebugHooks()` - opt-in diagnostic-flavored console output). `bytesToHex`/`hexToBytes` stay at the root. -## 2. Two input shapes now throw (transitional, one rc cycle) +## 2. Two input shapes now throw These previously mapped to tagged values SILENTLY; changing that quietly would corrupt bytes, so they throw a directive `CborError` (code `Custom`): @@ -246,18 +246,3 @@ precedent). Tagged types keep `cborTags()` / `untaggedCbor()` / The prefix grammar is policy (CONTRIBUTING.md): `is*` = narrowing guard, `as*` = `T | undefined`, `expect*` = `T` or throw `CborError`, `try*` = returns `Result`, never throws. - -## 11. What did NOT change - -The wire format (byte-for-byte, including every decoder rejection and its -error code); the accessor names that already followed the grammar -(`isMap`, `asText`, `expectArray`, `hasTag`, `getTaggedContent`, -`expectTaggedContent`, `tagValue`, `tagContent`, `arrayItem`, `arrayLength`, -`mapKeys`, `mapValues`, `mapSize`, …); `extractCbor` (now typed -`CborNative`); the bignum functions (tags 2/3); `encodeVarInt`/`decodeVarInt`; -`sortArrayByCborEncoding`; `cborEquals`; `bytesToHex`/`hexToBytes` names; -`getGlobalTagsStore` and all lookup names; the standard `TAG_*` constants; -`CborDate`'s `fromYmd`/`fromYmdHms`/`fromString`/`now`/`add`/`subtract`/ -`difference`/`equals`/`compare`/`toString`/`toJSON`; `CborSet`'s algebra -(`union`/`intersection`/`difference`/`isSubsetOf`/`isSupersetOf`); and -`walk`'s state-cloning visitor semantics (`@beta`). diff --git a/README.md b/README.md index 53afee2..3f1f964 100644 --- a/README.md +++ b/README.md @@ -59,6 +59,7 @@ Runnable examples live in the [`examples/`](https://github.com/BlockchainCommons ### Version History +- **Unreleased** - Closes every observable difference from `dcbor` 0.25.2: decoding keeps a leading U+FEFF; `cbor(string)` keeps the string and the encoder normalizes to NFC; `InvalidUtf8` messages mirror `core::str::Utf8Error`; float heads follow the reference's canonicality predicates (a whole f32 head at or beyond 2^31 decodes to an integer); float diagnostics round exact decimal ties like Rust; `CborDate` holds seconds plus nanoseconds (leap seconds display as `:60`, `NaN` is the epoch); `WrongTag` names both tags; `cborEquals` is structural; `registerStandardTags` registers unconditionally, tags are frozen and the global store is one per process across ESM and CJS; `expectUnsigned` gains fixed-width extraction and `TagsStore.clone()` is added. The Rust harness runs two builds over encode, decode (with messages), format, date and unsigned vectors, in CI. See `CHANGELOG.md` and `RUST_DIVERGENCES.md`. - **1.0.0-beta.2 (September 13, 2026)** - Diagnostic line breaking measures strings in UTF-8 bytes as the reference does; dates outside the reference's representable range are `InvalidDate` (an integer `f64` cannot hold is `OutOfRange` on decode) instead of a late `RangeError`; `registerStandardTags` names the bignum tags only on request, as the reference names them only under `num-bigint`; a tag-registration conflict is a `CborError`; `CborDate.fromString` parses exactly as `Date::from_string` (nanosecond fractions, leap seconds, chrono's bare-date forms), the component constructors validate instead of rolling over, and `toString()` prints years outside 0–9999 as the reference does. - **1.0.0-beta.1 (July 21, 2026)** - Initial beta implementation. diff --git a/RUST_DIVERGENCES.md b/RUST_DIVERGENCES.md index d59df3b..103f30c 100644 --- a/RUST_DIVERGENCES.md +++ b/RUST_DIVERGENCES.md @@ -1,43 +1,71 @@ # Divergences from the Rust reference implementation `@blockchaincommons/dcbor` is a port of [bc-dcbor-rust](https://github.com/BlockchainCommons/bc-dcbor-rust) -(crate `dcbor`). Its committed golden wire vectors (`tests/vectors/*.json`) -have been cross-validated against the Rust reference, **pinned at -`dcbor = 0.25.2`**, using the harness in `tests/rust-validation/`: +(crate `dcbor`). Its committed vectors (`tests/vectors/*.json`) are +cross-validated against the Rust reference, **pinned at `dcbor = 0.25.2`**, +by the harness in `tests/rust-validation/`. It runs twice: once against the +default build every Rust consumer uses, once with dcbor's `num-bigint` +feature. Both runs are a CI job. ```sh cd tests/rust-validation cargo run --release -- ../vectors +cargo run --release --features bignum -- ../vectors ``` -Validation result (2026-09-13): +Validation result (2026-09-14), default build: ``` -encode: 403 vectors - 357 match, 34 emulated-throw, 11 skipped (JS-only), 1 expected-divergence, 0 MISMATCH -decode: 241 vectors - 241 match, 0 expected-divergence, 0 MISMATCH +encode: 416 vectors - 348 match, 30 reference-throw, 8 emulated-throw, 30 skipped, 0 expected-divergence, 0 MISMATCH +decode: 267 vectors - 79 match, 188 reference-throw, 0 expected-divergence, 0 MISMATCH +format: 102 vectors - 94 match, 8 skipped, 0 expected-divergence, 0 MISMATCH +date: 60 vectors - 39 match, 13 reference-throw, 8 emulated-throw, 0 expected-divergence, 0 MISMATCH +uint: 19 vectors - 11 match, 8 reference-throw, 0 expected-divergence, 0 MISMATCH ``` -The compared wire vectors match except for the documented date case. All -177 decode rejections in this corpus match by error code, and the 34 -emulated throws are inputs both sides reject (the harness runs the -reference's constructor — `Date::from_string`, `BigUint` parsing — and -expects the port's error code). This verifies the selected fixtures, not -every possible input. Bare-float reduction quirks, negative zero, canonical -NaN, infinities, float-width selection, and the date-string grammar are -covered by the corpus. - -The harness compares bytes and error codes. Format output — diagnostic -notation in its flat, annotated and summarised modes, and annotated hex — -was validated by execution against the reference over a 35-value probe -corpus during the 1.0.0-beta.2 review; the rows that differed (the -line-breaking threshold, which counts UTF-8 bytes on both sides) are pinned -in `tests/format.test.ts`, and since 1.0.0-beta.2 every row matches. -`CborDate.toString()` was validated the same way (§1.3). - -This document records known differences and JavaScript input mappings. -The date-vector exception is recorded in the -`expected_divergences()` allowlist in `tests/rust-validation/src/main.rs` - -keep the two in sync. +`--features bignum`: + +``` +encode: 416 vectors - 367 match, 30 reference-throw, 8 emulated-throw, 11 skipped, 0 expected-divergence, 0 MISMATCH +decode: 267 vectors - 79 match, 188 reference-throw, 0 expected-divergence, 0 MISMATCH +format: 102 vectors - 99 match, 3 skipped, 0 expected-divergence, 0 MISMATCH +date: 60 vectors - 39 match, 13 reference-throw, 8 emulated-throw, 0 expected-divergence, 0 MISMATCH +uint: 19 vectors - 11 match, 8 reference-throw, 0 expected-divergence, 0 MISMATCH +``` + +- Encoded bytes match byte for byte, and every decode rejection matches by + error code **and** by the error's `Display` message (a `reference-throw` + on the decode, date and uint lines is a rejection both sides produce). +- **reference-throw** counts inputs the reference itself rejects + (`Date::from_string`, decoders). **emulated-throw** counts TypeScript + guards for inputs the reference cannot express (`cbor(bigint)` outside + the CBOR integer range, a negative `biguintToCbor`) or on which it panics + (`fromEpochSeconds(±Infinity)`, whole seconds outside chrono's range); the + harness probes chrono's `timestamp_opt` with the reference's own + arithmetic before calling into it. +- **skipped** counts JS-only input shapes (`Symbol`, function, malformed + bare node), the rejected legacy protocol shapes (`{tag, value}` literals + and `taggedCbor()`-only objects), rows pinned to the other build, and the + bignum recipes in the default build. +- Decode vectors whose accepted input re-encodes differently (whole-valued + f32/f64 heads) carry the reference's re-encoded bytes. +- Format vectors cover diagnostic notation (plain, annotated, flat, + summarized) and annotated hex under `TagsStoreOpt::None` and a fresh store + after `register_tags_in`. Each row that depends on the build is pinned to it. +- Date vectors cover `CborDate` decoding and display, including leap seconds, + the range bounds and `NaN`. +- Unsigned vectors compare `u8`/`u16`/`u32`/`u64::try_from` with + `expectUnsigned(cbor, { width, wrapNegative: true })` (§1.1). +- The allowlist `expected_divergences()` in `tests/rust-validation/src/main.rs` + is empty. +- This verifies the committed vectors, not every input. The entries below + are the behavioral differences that remain. + +Where the reference panics (an impossible date, a tag registered without a +name or under a conflicting name), the port throws a `CborError` at the same +call. Where the outcome depends on the host rather than the port (recursion +depth, clock resolution, cloning of visitor state), neither side defines a +contract. Neither kind is a divergence and neither is recorded here. --- @@ -46,120 +74,51 @@ keep the two in sync. These cases differ in accepted inputs or error handling. Changes should be reviewed against both the TypeScript tests and the Rust reference. -### 1.1 Non-finite and out-of-range date timestamps: TS guards, Rust saturates or panics - -| input | @blockchaincommons/dcbor | dcbor (Rust) | -|---|---|---| -| `CborDate.fromEpochSeconds(NaN)` (`from_timestamp`) | throws `InvalidDate` ("non-finite timestamp") | saturating-casts NaN to epoch `0`, encodes `c100` | -| `CborDate.fromEpochSeconds(±Infinity)` | throws `InvalidDate` | **panics** (`No such local time`, `date.rs:191`: `timestamp_opt(…).unwrap()`) | -| a finite timestamp outside chrono's range `[-8334601228800, 8210266876799]` s, at construction (`fromEpochSeconds`, `fromDate`) or decode (`c11b000007779a0a6b80`) | throws `InvalidDate` ("timestamp outside the representable range") | **panics** (`from_timestamp`); a chrono value given to `from_datetime` cannot be outside it | -| `CborDate.fromDate(new Date(NaN))` (`from_datetime`) | throws `InvalidDate` ("non-finite timestamp") | no analog: a chrono `DateTime` is always valid | -| impossible components — `fromYmd(2023, 13, 1)`, `fromYmdHms(…, 10, 30, 60)`, a year beyond ±262143 (`from_ymd`, `from_ymd_hms`) | throws `InvalidDate` ("Invalid date components") | **panics** (`with_ymd_and_hms(…).unwrap()`) | -| decode of an integer timestamp `f64` cannot hold exactly (`c11b7fffffffffffffff`) | throws `OutOfRange` | `OutOfRange` (`f64::exact_from_u64`) — identical | - -**Why:** the reference's `Date::from_timestamp` does `trunc() as i64` / -`fract() * 1e9 as u32` (saturating casts) and then `unwrap`s chrono's -`timestamp_opt`, so only NaN yields a (wrong) date and everything else -outside the range panics. The port rejects all of it with its -own `InvalidDate` at the same point (construction or decode) and reports the -reference's `OutOfRange` where the reference does; every accepted value also -renders (`toString()` never throws). Executed on the reference at the exact -bounds (`+262142-12-31T23:59:59Z` and `-262143-01-01` decode on both sides; -one second beyond panics there). A fallible Rust constructor would allow callers to handle these invalid inputs. - -**Affected vectors:** `date/non-finite-throws` (1 encode vector). The -differential corpus (`tests/corpus/corpus.ts`) carries no date recipes. - ---- +### 1.1 Negative integers converted to unsigned -### 1.2 Tag registration errors +The reference's `u8`…`u64` `TryFrom` wraps a negative integer instead +of rejecting it (`int.rs`, `Ok((-1 - a) as $type)`; executed: +`u8::try_from(-1) = 255`, `u64::try_from(-2^64) = 0`). +`expectUnsigned(cbor, { width, wrapNegative: true })` matches +`u8`…`u64`/`usize::try_from`, including the wrap and the `OutOfRange` +beyond the width (harness `uint` vectors). Without options, +`expectUnsigned` still throws `WrongType` for any negative integer. A +consumer that ports a field decoded with `u*::try_from` records whether it +matches the wrap: bc-components-ts uses the helper, and bc-known-values-ts +records its choice. -An unnamed tag or a tag value re-registered with a different name throws -`CborError` with code `Custom` in TypeScript. Rust uses an assertion or panic. -Both reject the registration, but the error mechanism differs. This is covered -by `tests/tags-store.test.ts`, not by the wire-vector allowlist. +### 1.2 Hex input tolerates whitespace -### 1.3 A parsed leap second displays differently until it is encoded +`hexToBytes` strips whitespace before decoding, so `"a1 61 61 01"` yields +the bytes of `"a1616101"`; an odd length or a non-hex digit throws +`CborError` `Custom`. The reference's `CBOR::try_from_hex` unwraps +`hex::decode`, which rejects whitespace, so it panics on every malformed +input, whitespace included. The port therefore accepts an input the +reference does not. Covered by `tests/hex.property.test.ts`. -`CborDate.fromString("2023-12-25T10:30:60Z").toString()` prints -`2023-12-25T10:31:00Z`; the reference's `Date` prints `2023-12-25T10:30:60Z`. -chrono keeps the leap second as second 59 plus a second of nanoseconds, -while the port stores the timestamp — which is what both sides encode -(`c11a658959e4`, identical). After a CBOR round trip the reference prints -`10:31:00Z` too (executed). Every other `Display` row matches: `%Y-%m-%d` -when the clock reads 00:00:00 (a fraction of a second does not count), -otherwise RFC 3339 to the second, years outside 0–9999 with a sign and at -least four digits (`-0004-02-29`, `+12023-02-08`); pinned in -`tests/date.test.ts`. - -## 2. JS-only input domain (no Rust analog exists) - -These inputs cannot be expressed against the Rust API at all, so there is -nothing to compare - the TS behavior is frozen by the golden vectors alone. -The Rust harness **skips** them. - -| input shape | frozen TS behavior | vector | -|---|---|---| -| `Symbol` passed to `cbor()` | throws `CborError` `Custom` ("Unsupported type for CBOR encoding") | `unsupported/symbol-throws` | -| function passed to `cbor()` | throws `Custom` | `unsupported/function-throws` | -| malformed bare Cbor node (`{isCbor: true, type: ByteString, value: 42}`) | throws `WrongType` at encode | `rawbad/malformed-bytestring-node-throws` | +--- -Section 3 describes additional input mappings and rejected legacy protocol -shapes; the rejected legacy shapes are also skipped by the harness. +## 2. Reference quirks the port reproduces ---- +### 2.1 `i16::exact_from_f16` excludes -32768 -## 3. Mapping equivalences (JS-specific inputs validated via their byte-target) - -These are JS-specific input *routes* whose output bytes were validated -against the equivalent Rust construction. Where a row has a Rust byte target, the corpus compares those bytes. -Rows documenting rejected legacy inputs are TypeScript-only checks. - -| JS-specific input | maps to (validated against Rust) | notes / vector | -|---|---|---| -| `cbor(undefined)` | CBOR `null` (`f6`) | Rust has no `undefined`; dCBOR forbids simple 23 (`f7`) - both decoders reject it. `simple/undefined-maps-to-null` | -| lone surrogate strings (`"\ud800"`) | U+FFFD replacement (`63efbfbd` for the 3-byte text) | JS `TextEncoder` replacement semantics; Rust `String` cannot hold lone surrogates. `str/lone-surrogate-becomes-replacement` | -| JS `Set` input | plain array in **insertion order** (SameValueZero dedup) | unlike `CborSet` (canonical sort + dedup), which matches Rust `Set` exactly. `jsset/*` | -| JS `Map` / plain-object input | `CborMap` → canonical key-byte order, duplicate canonical keys last-write-wins | Rust `Map::insert` behaves identically once constructed. `jsmap/*`, `obj/*` | -| `{tag: T, value: V}` two-own-key literal | **throws directive `Custom`** (was: `to_tagged_value(T, V)` bytes) | TS-only tombstone; skipped by the Rust harness. `tagobjlit/*` | -| objects with inherited `tag`/`value` (prototype) | plain-object→map of OWN keys only | freezes the sniffing arm's own-keys boundary. `protoobj/*` | -| `toCbor()` protocol objects | the underlying value's bytes | the ONE encode protocol. `tocbor/*` | -| `taggedCbor()`-only objects | **throws directive `Custom`** (was: auto-wrapped) | TS-only tombstone; skipped by the Rust harness. `taggedproto/*` | -| objects with BOTH protocols | `["toCbor-won", inner]` bytes - **`toCbor` wins** (was: `taggedCbor` won) | the harness mirrors the new precedence. `bothproto/*` | -| `cbor(bigint)` outside `[-(2⁶⁴), 2⁶⁴−1]` | throws `OutOfRange` | Rust cannot express this input: it has **no** `i128`/`u128 → CBOR` conversion (its own tests construct via `CBORCase`), so the range guard is TS surface behavior. The in-range bigint bytes match Rust's `CBORCase::Unsigned`/`Negative` exactly. `int/2^64-bigint-throws`, `int/below-cbor-int-min-throws` | -| `biguintToCbor(negative)` | throws `OutOfRange` | Rust's `From` cannot receive a negative - type-level in Rust, runtime guard in TS. `biguint/negative-throws` | -| number vs bigint input forms | JS `number` follows Rust's `From` semantics; JS `bigint` follows `CBORCase` integer construction | e.g. the **number** literal `18446744073709551615` is the double 2⁶⁴ → float `fa5f800000` in both (`From` parity), while `18446744073709551615n` → `1bffffffffffffffff`. `int/u64-max-as-number-is-float` | - -`CborDate.fromString` was validated directly against Rust's -`Date::from_string` by execution (66 strings, byte-identical outcomes; the -`datestr/*` vectors run each form through the reference): the RFC 3339 form -keeps nine fraction digits and the `:60` leap second, takes `T`/`t`/space -and `−` (U+2212) in the offset, and bounds the offset by ±23:59; the bare -form is chrono's `%Y-%m-%d` (one- or two-digit month and day, a signed year -of any length, whitespace before each number); both reject the same -malformed inputs (`datestr/invalid-throws`, `datestr/missing-offset-throws` -→ `InvalidDate`). The stored timestamp is the reference's `timestamp()` -arithmetic — whole seconds plus nanoseconds over 10⁹ — and `fromDate` -computes a JS `Date`'s milliseconds the same way, so a fraction encodes to -the same bytes as `Date::from_datetime`. The impossible-component and -leap-second-display cases are §1.1 and §1.3. - -- **Standard tags 2 and 3.** The reference names `positive-bignum` / - `negative-bignum` (and summarises `bignum(…)`) only when built with its - `num-bigint` feature; when that feature is disabled, a Rust peer prints `2(h'…')` and annotates `# tag(2)`. Since - 1.0.0-beta.2 `registerStandardTags(store)` matches that, and - `registerStandardTags(store, { bignum: true })` matches the `num-bigint` - build (executed: identical output in both configurations). +The reference's `exact_from_f16` for `i16` tests `source <= -32768.0` +(`exact.rs`), so `-32768.0`, which binary16 represents exactly, is `None`, +while `exact_from_f32(-32768.0)` and `exact_from_f64(-32768.0)` are +`Some(-32768)`. `ExactI16.exactFromF16(-32768)` is `undefined` to match; +`tests/exact.test.ts` pins it. Nothing on the wire depends on it. --- ## Maintenance -- **Adding vectors:** if a new vector diverges from Rust, either it is a bug - (fix it) or it belongs in one of the classes above - document it here and, when the wire harness covers it, - add it to `expected_divergences()` in `tests/rust-validation/src/main.rs` - in the same change. -- **Re-run the cross-validation** after every fixture regeneration - (`bun run vectors:generate`) and update this document when the result changes. +- **Adding vectors:** a vector that diverges from Rust is a bug to fix, + unless it belongs to a class above. Document such a case here and, when + the harness covers it, add it to `expected_divergences()` in + `tests/rust-validation/src/main.rs` in the same change. The allowlist is + empty today. +- **Re-run the cross-validation** in both builds (default and + `--features bignum`) after every fixture regeneration + (`bun run vectors:generate`), and update the header result lines. - **Version pin:** the harness pins `dcbor = 0.25.2`. When bumping, re-run - and update the header of this file with the new result line. + both builds and update the header. diff --git a/api/dcbor.api.md b/api/dcbor.api.md index 97c4f12..05f1403 100644 --- a/api/dcbor.api.md +++ b/api/dcbor.api.md @@ -115,7 +115,7 @@ export interface CborCodec { readonly tags?: readonly Tag[] | undefined; } -// @public (undocumented) +// @public export class CborDate implements CborTagged { get [Symbol.toStringTag](): string; add(seconds: number): CborDate; @@ -256,9 +256,13 @@ export class CborMap { clear(): void; // (undocumented) delete(key: CborInput): boolean; + // @internal + encodedKeyAt(i: number): Uint8Array; entries(): Generator<[Cbor, Cbor], void, undefined>; // @internal get entriesArray(): MapEntry[]; + // @internal + entryAt(i: number): MapEntry; forEach(callback: (value: Cbor, key: Cbor, map: CborMap) => void, thisArg?: unknown): void; get(key: CborInput): Cbor | undefined; getOrThrow(key: CborInput): Cbor; @@ -370,6 +374,7 @@ export interface CborTaggedType { readonly isCbor: true; // (undocumented) readonly tag: CborNumber; + readonly tagName?: string | undefined; // (undocumented) readonly type: typeof MajorType.Tagged; // (undocumented) @@ -425,7 +430,7 @@ export function decodeWith(data: Uint8Array, codec: CborCodec): T; // @public export const encodeCbor: (value: CborInput) => Uint8Array; -// @public (undocumented) +// @public export const encodeVarInt: (value: CborNumber, majorType: MajorType) => Uint8Array; // @public @@ -456,13 +461,19 @@ export const expectNegative: (cbor: Cbor) => number | bigint; export const expectNumber: (cbor: Cbor) => CborNumber; // @public -export const expectTaggedContent: (cbor: Cbor, tag: number | bigint) => Cbor; +export const expectTaggedContent: (cbor: Cbor, tag: number | bigint | Tag) => Cbor; // @public export const expectText: (cbor: Cbor) => string; // @public -export const expectUnsigned: (cbor: Cbor) => number | bigint; +export const expectUnsigned: (cbor: Cbor, options?: ExpectUnsignedOptions) => number | bigint; + +// @public +export interface ExpectUnsignedOptions { + readonly width: 8 | 16 | 32 | 64; + readonly wrapNegative?: boolean | undefined; +} // @public export const extractCbor: (cbor: Cbor | Uint8Array) => CborNative; @@ -602,7 +613,7 @@ export interface ReadonlyTagsStore { tagForValue(value: CborNumber): Tag | undefined; } -// @public (undocumented) +// @public export const registerStandardTags: (store?: TagsStore, options?: RegisterStandardTagsOptions) => void; // @public @@ -662,48 +673,9 @@ export const Tag: { }; // @public -export const TAG_BASE16 = 23; - -// @public -export const TAG_BASE64 = 22; - -// @public -export const TAG_BASE64_TEXT = 34; - -// @public -export const TAG_BASE64URL = 21; - -// @public -export const TAG_BASE64URL_TEXT = 33; - -// @public -export const TAG_BIGFLOAT = 5; - -// @public -export const TAG_BINARY_UUID = 257; - -// @public (undocumented) export const TAG_DATE = 1; // @public -export const TAG_DATE_TIME_STRING = 0; - -// @public -export const TAG_DECIMAL_FRACTION = 4; - -// @public -export const TAG_ENCODED_CBOR = 24; - -// @public -export const TAG_EPOCH_DATE = 100; - -// @public -export const TAG_EPOCH_DATE_TIME = 1; - -// @public -export const TAG_MIME_MESSAGE = 36; - -// @public (undocumented) export const TAG_NAME_DATE = "date"; // @public @@ -718,24 +690,6 @@ export const TAG_NEGATIVE_BIGNUM = 3; // @public export const TAG_POSITIVE_BIGNUM = 2; -// @public -export const TAG_REGEXP = 35; - -// @public -export const TAG_SELF_DESCRIBE_CBOR = 55799; - -// @public -export const TAG_SET = 258; - -// @public -export const TAG_STRING_REF_NAMESPACE = 256; - -// @public -export const TAG_URI = 32; - -// @public -export const TAG_UUID = 37; - // @public export const tagContent: (cbor: Cbor) => Cbor | undefined; @@ -751,12 +705,13 @@ export class TagsStore implements ReadonlyTagsStore { constructor(); // (undocumented) assignedNameForTag(tag: Tag): string | undefined; + clone(): TagsStore; // (undocumented) nameForTag(tag: Tag): string; // (undocumented) nameForValue(value: CborNumber): string; register(tag: Tag): void; - registerAll(tags: Tag[]): void; + registerAll(tags: Iterable): void; setSummarizer(tagValue: CborNumber, summarizer: CborSummarizer): void; // (undocumented) summarizer(tag: CborNumber): CborSummarizer | undefined; diff --git a/api/diagnostic.d.mts b/api/diagnostic.d.mts index 1ea4700..66c2a29 100644 --- a/api/diagnostic.d.mts +++ b/api/diagnostic.d.mts @@ -1,6 +1,6 @@ -import { t as Cbor } from "./cbor-D4SlBSmQ.mjs"; -import { a as TagsStoreOpt, i as TagsStore } from "./tags-store-D853hYTA.mjs"; -import { i as WalkElement } from "./walk-Te23BaT5.mjs"; +import { t as Cbor } from "./cbor-CZgweoMn.mjs"; +import { a as TagsStoreOpt, i as TagsStore } from "./tags-store-8mi5mOuA.mjs"; +import { i as WalkElement } from "./walk-B0aveFey.mjs"; //#region src/diag.d.ts /** * Options for diagnostic formatting. @@ -8,14 +8,15 @@ import { i as WalkElement } from "./walk-Te23BaT5.mjs"; interface DiagFormatOpts { /** * Add tag names as annotations. - * When true, tagged values are displayed as "tagName(content)" instead of "tagValue(content)". + * When true, a tagged value whose tag has a name in the store is followed by + * a `/ name /` comment, e.g. `1(1675854714) / date /`. * * @default false */ annotate?: boolean | undefined; /** * Use custom summarizers for tagged values. - * When true, calls registered summarizers for tagged values. + * When true, calls registered summarizers for tagged values. Implies `flat`. * * @default false */ diff --git a/api/index.d.mts b/api/index.d.mts index f284b60..4c7faf8 100644 --- a/api/index.d.mts +++ b/api/index.d.mts @@ -1,5 +1,5 @@ -import { A as CborMap, C as CborDate, D as extractTaggedContent, E as decodeWith, F as isCborNaN, I as simpleName, M as CborNative, N as extractCbor, O as validateTag, P as Simple, S as isCborNumber, T as CborTagged, _ as CborTextType, a as taggedValue, b as ToCbor, c as CborArrayType, d as CborMapType, f as CborMethods, g as CborTaggedType, h as CborSimpleType, i as encodeCbor, j as MapEntry, k as ByteString, l as CborByteStringType, m as CborNumber, n as cbor, o as Tag, p as CborNegativeType, r as cborEquals, s as TagValue, t as Cbor, u as CborInput, v as CborUnsignedType, w as CborCodec, x as isCbor, y as MajorType } from "./cbor-D4SlBSmQ.mjs"; -import { a as TagsStoreOpt, c as CborError, d as CborErrorDetailsByCode, f as CborErrorTyped, h as Result, i as TagsStore, l as CborErrorCode, m as Ok, n as ReadonlyTagsStore, o as getGlobalTagsStore, p as Err, r as SummarizerResult, s as withTags, t as CborSummarizer, u as CborErrorDetails } from "./tags-store-D853hYTA.mjs"; +import { A as CborMap, C as CborDate, D as extractTaggedContent, E as decodeWith, F as isCborNaN, I as simpleName, M as CborNative, N as extractCbor, O as validateTag, P as Simple, S as isCborNumber, T as CborTagged, _ as CborTextType, a as taggedValue, b as ToCbor, c as CborArrayType, d as CborMapType, f as CborMethods, g as CborTaggedType, h as CborSimpleType, i as encodeCbor, j as MapEntry, k as ByteString, l as CborByteStringType, m as CborNumber, n as cbor, o as Tag, p as CborNegativeType, r as cborEquals, s as TagValue, t as Cbor, u as CborInput, v as CborUnsignedType, w as CborCodec, x as isCbor, y as MajorType } from "./cbor-CZgweoMn.mjs"; +import { a as TagsStoreOpt, c as CborError, d as CborErrorDetailsByCode, f as CborErrorTyped, h as Result, i as TagsStore, l as CborErrorCode, m as Ok, n as ReadonlyTagsStore, o as getGlobalTagsStore, p as Err, r as SummarizerResult, s as withTags, t as CborSummarizer, u as CborErrorDetails } from "./tags-store-8mi5mOuA.mjs"; //#region src/hex.d.ts /** * Convert bytes to a lowercase hex string. @@ -42,14 +42,14 @@ export declare const hexToBytes: (hexString: string) => Uint8Array; * @remarks Decoded byte strings are zero-copy views aliasing the input * buffer - mutating the input after decoding (or mutating the returned * bytes) changes the other side. Call `.slice()` first if you need an - * independent copy. This is deliberate: the zero-copy decode performance - * profile is part of the library's contract. + * independent copy. */ export declare function decodeCbor(data: Uint8Array): Cbor; /** * Decode without throwing: returns a {@link Result} carrying the decoded value, * or the {@link CborError} that {@link decodeCbor} would have thrown. Non-CBOR - * errors still propagate. + * errors still propagate - including the host's `RangeError` when a deeply + * nested input exhausts the call stack (the reference aborts there too). * * The `try` prefix means "returns `Result`, never throws" - everywhere in * this library. @@ -88,9 +88,8 @@ export declare class CborSet { * Create a CborSet from any iterable of encodable items. Duplicates (by * canonical encoding) are removed. * - * NOTE: strings are iterable - `CborSet.from("abc")` is a THREE-element - * set of one-character strings, not a single-element set. Wrap in an - * array (`CborSet.from(["abc"])`) for the latter. + * Strings are iterable: `CborSet.from("abc")` is a three-element set of + * one-character strings. Use `CborSet.from(["abc"])` for a single element. */ static from(items: Iterable): CborSet; /** @@ -141,10 +140,8 @@ export declare class CborSet { isSupersetOf(other: CborSet): boolean; [Symbol.iterator](): Generator; /** - * Iterate the stored `Cbor` elements lazily in canonical order. - * - * NOTE: this yields the stored `Cbor` nodes. Use `toArray()` for an eager - * array of extracted native values. + * Iterate the stored `Cbor` elements lazily in canonical order. Use + * `toArray()` for an eager array of extracted native values. */ values(): Generator; /** JS `Set.keys()` mirror - identical to `values()`. */ @@ -192,24 +189,12 @@ export declare class CborSet { } //#endregion //#region src/tags.d.ts -/** - * Tag 0: Standard date/time string (RFC 3339) - */ -export declare const TAG_DATE_TIME_STRING = 0; -/** - * Tag 1: Epoch-based date/time (seconds since 1970-01-01T00:00:00Z) - */ -export declare const TAG_EPOCH_DATE_TIME = 1; -/** - * Tag 100: Epoch-based date (days since 1970-01-01) - */ -export declare const TAG_EPOCH_DATE = 100; /** * Tag 2: Positive bignum (unsigned arbitrary-precision integer) */ export declare const TAG_POSITIVE_BIGNUM = 2; /** - * Tag 3: Negative bignum (signed arbitrary-precision integer) + * Tag 3: Negative bignum (arbitrary-precision negative integer) */ export declare const TAG_NEGATIVE_BIGNUM = 3; /** @@ -221,81 +206,13 @@ export declare const TAG_NAME_POSITIVE_BIGNUM = "positive-bignum"; */ export declare const TAG_NAME_NEGATIVE_BIGNUM = "negative-bignum"; /** - * Tag 4: Decimal fraction [exponent, mantissa] - */ -export declare const TAG_DECIMAL_FRACTION = 4; -/** - * Tag 5: Bigfloat [exponent, mantissa] - */ -export declare const TAG_BIGFLOAT = 5; -/** - * Tag 21: Expected conversion to base64url encoding - */ -export declare const TAG_BASE64URL = 21; -/** - * Tag 22: Expected conversion to base64 encoding - */ -export declare const TAG_BASE64 = 22; -/** - * Tag 23: Expected conversion to base16 encoding - */ -export declare const TAG_BASE16 = 23; -/** - * Tag 24: Encoded CBOR data item - */ -export declare const TAG_ENCODED_CBOR = 24; -/** - * Tag 32: URI (text string) - */ -export declare const TAG_URI = 32; -/** - * Tag 33: base64url-encoded text - */ -export declare const TAG_BASE64URL_TEXT = 33; -/** - * Tag 34: base64-encoded text - */ -export declare const TAG_BASE64_TEXT = 34; -/** - * Tag 35: Regular expression (PCRE/ECMA262) - */ -export declare const TAG_REGEXP = 35; -/** - * Tag 36: MIME message - */ -export declare const TAG_MIME_MESSAGE = 36; -/** - * Tag 37: Binary UUID - */ -export declare const TAG_UUID = 37; -/** - * Tag 256: string reference (namespace) - */ -export declare const TAG_STRING_REF_NAMESPACE = 256; -/** - * Tag 257: binary UUID reference - */ -export declare const TAG_BINARY_UUID = 257; -/** - * Tag 258: Set of values (array with no duplicates) - */ -export declare const TAG_SET = 258; -/** - * Tag 55799: Self-describe CBOR (magic number 0xd9d9f7) + * Tag 1: Epoch-based date/time (seconds since 1970-01-01T00:00:00Z) */ -export declare const TAG_SELF_DESCRIBE_CBOR = 55799; export declare const TAG_DATE = 1; -export declare const TAG_NAME_DATE = "date"; /** - * Register the standard tags (date, bignums) and their summarizers into - * `store`. - * - * Idempotent: tags already registered under the same name are skipped; - * registering a value under a DIFFERENT name still throws via the store's - * conflict validation. - * - * @param store - Target store; defaults to the global tags store. + * Name for tag 1 (date). */ +export declare const TAG_NAME_DATE = "date"; /** Options for {@link registerStandardTags}. */ interface RegisterStandardTagsOptions { /** @@ -307,33 +224,29 @@ interface RegisterStandardTagsOptions { */ readonly bignum?: boolean | undefined; } -export declare const registerStandardTags: (store?: TagsStore, options?: RegisterStandardTagsOptions) => void; /** - * Converts an array of tag values to their corresponding Tag objects. + * Register the standard tags (date, and the bignums with `bignum`) and their + * summarizers into `store`. * - * This function looks up each tag value in the global tag registry and returns - * an array of complete Tag objects. For any tag values that aren't - * registered in the global registry, it creates a basic Tag with just the - * value (no name). + * Re-registering is idempotent and moves each standard name back to its + * standard value, as the reference's `insert_all` does: a store that had + * named tag 99 `date` names tag 1 `date` afterwards. Registering tag 1 (or + * 2/3 with `bignum`) under a different name throws `CborError` `Custom` + * from the store's conflict validation, before any summarizer is set. * - * @param values - Array of numeric tag values to convert - * @returns Array of Tag objects corresponding to the input values + * @param store - Target store; defaults to the global tags store. + */ +export declare const registerStandardTags: (store?: TagsStore, options?: RegisterStandardTagsOptions) => void; +/** + * Resolve tag values through the global tags store. A value the store does + * not know becomes an unnamed `Tag`. * * @example * ```typescript - * // Register some tags first * registerStandardTags(); - * - * // Convert tag values to Tag objects - * const tags = tagsForValues([1, 42, 999]); - * - * // The first tag (value 1) should be registered as "date" - * console.log(tags[0].value); // 1 - * console.log(tags[0].name); // "date" - * - * // Unregistered tags will have a value but no name - * console.log(tags[1].value); // 42 - * console.log(tags[2].value); // 999 + * const tags = tagsForValues([1, 42]); + * tags[0].name; // "date" + * tags[1].name; // undefined * ``` */ export declare const tagsForValues: (values: (number | bigint)[]) => Tag[]; @@ -347,7 +260,7 @@ export declare const tagsForValues: (values: (number | bigint)[]) => Tag[]; * * @param value - A non-negative bigint (must be >= 0n) * @returns CBOR tagged value - * @throws CborError with type OutOfRange if value is negative + * @throws {CborError} `OutOfRange` if value is negative */ export declare function biguintToCbor(value: bigint): Cbor; /** @@ -372,8 +285,8 @@ export declare function bigintToCbor(value: bigint): Cbor; * * @param cbor - A CBOR value that should be a byte string * @returns Non-negative bigint - * @throws CborError with type WrongType if not a byte string - * @throws CborError with type NonCanonicalNumeric if encoding is non-canonical + * @throws {CborError} `WrongType` if not a byte string + * @throws {CborError} `NonCanonicalNumeric` if encoding is non-canonical */ export declare function biguintFromUntaggedCbor(cbor: Cbor): bigint; /** @@ -388,8 +301,8 @@ export declare function biguintFromUntaggedCbor(cbor: Cbor): bigint; * * @param cbor - A CBOR value that should be a byte string * @returns Negative bigint - * @throws CborError with type WrongType if not a byte string - * @throws CborError with type NonCanonicalNumeric if encoding is non-canonical + * @throws {CborError} `WrongType` if not a byte string + * @throws {CborError} `NonCanonicalNumeric` if encoding is non-canonical */ export declare function bigintFromNegativeUntaggedCbor(cbor: Cbor): bigint; /** @@ -441,8 +354,8 @@ export declare function cborToBigint(cbor: Cbor): bigint; */ export declare function sortArrayByCborEncoding(array: readonly T[]): T[]; /** - * Sortable-by-CBOR-encoding interface shape. The `arraySortable` / - * `setSortable` helpers wrap any iterable into a `CBORSortable` view. + * A collection that can be sorted by CBOR encoding. `arraySortable` and + * `setSortable` wrap an array or a set in this interface. */ interface CBORSortable { sortByCborEncoding(): T[]; @@ -463,6 +376,12 @@ export declare function setSortable(set: ReadonlySet): C export declare const hasFractionalPart: (n: number) => boolean; //#endregion //#region src/varint.d.ts +/** + * Encode a CBOR head (major type + argument) in its shortest form. + * + * @throws {CborError} `OutOfRange` for a negative, fractional, or + * above-u64 argument. + */ export declare const encodeVarInt: (value: CborNumber, majorType: MajorType) => Uint8Array; export declare const decodeVarIntData: (dataView: DataView, offset: number) => { majorType: MajorType; @@ -617,9 +536,7 @@ export declare const asInteger: (cbor: Cbor) => number | bigint | undefined; * * Decoded byte strings are zero-copy views aliasing the input buffer - * mutating the input after decoding (or mutating the returned bytes) changes - * the other side. Call `.slice()` first if you need an independent copy. This - * is deliberate: the zero-copy decode performance profile is part of the - * library's contract. + * the other side. Call `.slice()` first if you need an independent copy. * * @param cbor - CBOR value * @returns Byte string or undefined @@ -686,7 +603,7 @@ export declare const arrayLength: (cbor: Cbor) => number | undefined; * Check if array is empty. * * @param cbor - CBOR value (must be array) - * @returns True if empty, false if not empty, undefined if not array + * @returns True if empty; false if not empty or not an array (matching hasTag) */ export declare const arrayIsEmpty: (cbor: Cbor) => boolean; /** @@ -702,7 +619,7 @@ export declare function mapValue(cbor: Cbor, key: CborInput): Cbor | undefined; * * @param cbor - CBOR value (must be map) * @param key - Map key - * @returns True if key exists, false otherwise, undefined if not map + * @returns True if key exists; false otherwise or if not a map (matching hasTag) */ export declare function mapHas(cbor: Cbor, key: CborInput): boolean; /** @@ -730,7 +647,7 @@ export declare const mapSize: (cbor: Cbor) => number | undefined; * Check if map is empty. * * @param cbor - CBOR value (must be map) - * @returns True if empty, false if not empty, undefined if not map + * @returns True if empty; false if not empty or not a map (matching hasTag) */ export declare const mapIsEmpty: (cbor: Cbor) => boolean; /** @@ -772,20 +689,45 @@ export declare const getTaggedContent: (cbor: Cbor, tag: number | bigint) => Cbo export declare const asTaggedValue: (cbor: Cbor) => [Tag, Cbor] | undefined; //#endregion //#region src/conveniences-expect.d.ts +/** + * Options for {@link expectUnsigned}: extract into a fixed-width unsigned + * integer the way the reference's `u8`/`u16`/`u32`/`u64: TryFrom` + * do (dcbor 0.25.2 `int.rs`). + */ +interface ExpectUnsignedOptions { + /** Target width in bits; an unsigned value above 2^width − 1 is `OutOfRange`. */ + readonly width: 8 | 16 | 32 | 64; + /** + * Also accept a negative integer node and wrap it as the reference does: + * a value `v` in [−2^width, −1] yields `2^width + v` (so −1 is 255 at + * width 8), and a value below −2^width is `OutOfRange`. Off by default: + * without it a negative node is `WrongType`, as for every other type. See + * RUST_DIVERGENCES.md §1.1. + */ + readonly wrapNegative?: boolean | undefined; +} /** * Extract unsigned integer value, throwing if type doesn't match. * + * With `options`, the value is checked against a fixed width and, when + * `wrapNegative` is set, a negative node is wrapped exactly as the + * reference's `u*::try_from` wraps it (see {@link ExpectUnsignedOptions}). + * * @param cbor - CBOR value - * @returns Unsigned integer - * @throws {CborError} With type 'WrongType' if cbor is not an unsigned integer + * @param options - Fixed-width extraction (optional; without it the + * behaviour is the plain `Unsigned`-or-`WrongType` check) + * @returns Unsigned integer (`bigint` above `Number.MAX_SAFE_INTEGER`) + * @throws {CborError} `WrongType` if cbor is not an unsigned integer (or, + * with `wrapNegative`, not an integer); `OutOfRange` when the value does + * not fit `width` */ -export declare const expectUnsigned: (cbor: Cbor) => number | bigint; +export declare const expectUnsigned: (cbor: Cbor, options?: ExpectUnsignedOptions) => number | bigint; /** * Extract negative integer value, throwing if type doesn't match. * * @param cbor - CBOR value * @returns Negative integer - * @throws {CborError} With type 'WrongType' if cbor is not a negative integer + * @throws {CborError} `WrongType` if cbor is not a negative integer */ export declare const expectNegative: (cbor: Cbor) => number | bigint; /** @@ -793,21 +735,19 @@ export declare const expectNegative: (cbor: Cbor) => number | bigint; * * @param cbor - CBOR value * @returns Integer - * @throws {CborError} With type 'WrongType' if cbor is not an integer + * @throws {CborError} `WrongType` if cbor is not an integer */ export declare const expectInteger: (cbor: Cbor) => number | bigint; /** * Extract byte string value, throwing if type doesn't match. * + * Decoded byte strings are zero-copy views aliasing the input buffer - + * mutating the input after decoding (or mutating the returned bytes) changes + * the other side. Call `.slice()` first if you need an independent copy. + * * @param cbor - CBOR value * @returns Byte string - * @throws {CborError} With type 'WrongType' if cbor is not a byte string - * - * NOTE: decoded byte strings are zero-copy views aliasing the input - * buffer - mutating the input after decoding (or mutating the returned - * bytes) changes the other side. Call `.slice()` first if you need an - * independent copy. This is deliberate: the zero-copy decode performance - * profile is part of the library's contract. + * @throws {CborError} `WrongType` if cbor is not a byte string */ export declare const expectBytes: (cbor: Cbor) => Uint8Array; /** @@ -815,7 +755,7 @@ export declare const expectBytes: (cbor: Cbor) => Uint8Array; * * @param cbor - CBOR value * @returns Text string - * @throws {CborError} With type 'WrongType' if cbor is not a text string + * @throws {CborError} `WrongType` if cbor is not a text string */ export declare const expectText: (cbor: Cbor) => string; /** @@ -823,7 +763,7 @@ export declare const expectText: (cbor: Cbor) => string; * * @param cbor - CBOR value * @returns Array - * @throws {CborError} With type 'WrongType' if cbor is not an array + * @throws {CborError} `WrongType` if cbor is not an array */ export declare const expectArray: (cbor: Cbor) => readonly Cbor[]; /** @@ -831,7 +771,7 @@ export declare const expectArray: (cbor: Cbor) => readonly Cbor[]; * * @param cbor - CBOR value * @returns Map - * @throws {CborError} With type 'WrongType' if cbor is not a map + * @throws {CborError} `WrongType` if cbor is not a map */ export declare const expectMap: (cbor: Cbor) => CborMap; /** @@ -839,15 +779,18 @@ export declare const expectMap: (cbor: Cbor) => CborMap; * * @param cbor - CBOR value * @returns Boolean - * @throws {CborError} With type 'WrongType' if cbor is not a boolean + * @throws {CborError} `WrongType` if cbor is not a boolean */ export declare const expectBoolean: (cbor: Cbor) => boolean; /** * Extract float value, throwing if type doesn't match. * + * Integers coerce to float, as for {@link asFloat}. + * * @param cbor - CBOR value * @returns Float - * @throws {CborError} With type 'WrongType' if cbor is not a float + * @throws {CborError} `WrongType` if cbor is not numeric; `OutOfRange` if an + * integer is not exactly representable as f64 */ export declare const expectFloat: (cbor: Cbor) => number; /** @@ -855,20 +798,24 @@ export declare const expectFloat: (cbor: Cbor) => number; * * @param cbor - CBOR value * @returns Number - * @throws {CborError} With type 'WrongType' if cbor is not a number + * @throws {CborError} `WrongType` if cbor is not a number */ export declare const expectNumber: (cbor: Cbor) => CborNumber; /** - * Extract content if has specific tag, throwing if not. + * Extract content if has specific tag, throwing if not (the reference's + * `try_into_expected_tagged_value`). * - * Throws `{ type: "WrongType" }` if `cbor` is not tagged at all, otherwise - * `{ type: "WrongTag", expected, actual }` if the tag doesn't match. + * The `WrongTag` error names the expected tag as it was given (a `Tag` keeps + * its name; a number or bigint stays unnamed) and the actual tag as the node + * carries it. * * @param cbor - CBOR value - * @param tag - Expected tag value + * @param tag - Expected tag value, or a `Tag` * @returns Tagged content + * @throws {CborError} `WrongType` if `cbor` is not tagged; `WrongTag` (with + * `details.expectedTag` and `details.actualTag`) if the tag doesn't match */ -export declare const expectTaggedContent: (cbor: Cbor, tag: number | bigint) => Cbor; +export declare const expectTaggedContent: (cbor: Cbor, tag: number | bigint | Tag) => Cbor; //#endregion -export { ByteString, type CBORSortable, type Cbor, type CborArrayType, type CborByteStringType, type CborCodec, CborDate, CborError, type CborErrorCode, type CborErrorDetails, type CborErrorDetailsByCode, type CborErrorTyped, type CborInput, CborMap, type CborMapType, type CborMethods, type CborNative, type CborNegativeType, type CborNumber, type CborSimpleType, type CborSummarizer, type CborTagged, type CborTaggedType, type CborTextType, type CborUnsignedType, Err, MajorType, type MapEntry, Ok, type ReadonlyTagsStore, type RegisterStandardTagsOptions, type Result, type Simple, type SummarizerResult, Tag, type TagValue, TagsStore, type TagsStoreOpt, type ToCbor, cbor, cborEquals, decodeWith, encodeCbor, extractCbor, extractTaggedContent, getGlobalTagsStore, isCbor, isCborNaN, isCborNumber, simpleName, taggedValue, validateTag, withTags }; +export { ByteString, type CBORSortable, type Cbor, type CborArrayType, type CborByteStringType, type CborCodec, CborDate, CborError, type CborErrorCode, type CborErrorDetails, type CborErrorDetailsByCode, type CborErrorTyped, type CborInput, CborMap, type CborMapType, type CborMethods, type CborNative, type CborNegativeType, type CborNumber, type CborSimpleType, type CborSummarizer, type CborTagged, type CborTaggedType, type CborTextType, type CborUnsignedType, Err, type ExpectUnsignedOptions, MajorType, type MapEntry, Ok, type ReadonlyTagsStore, type RegisterStandardTagsOptions, type Result, type Simple, type SummarizerResult, Tag, type TagValue, TagsStore, type TagsStoreOpt, type ToCbor, cbor, cborEquals, decodeWith, encodeCbor, extractCbor, extractTaggedContent, getGlobalTagsStore, isCbor, isCborNaN, isCborNumber, simpleName, taggedValue, validateTag, withTags }; //# sourceMappingURL=index.d.mts.map \ No newline at end of file diff --git a/api/walk.d.mts b/api/walk.d.mts index 9df1d58..5b382d0 100644 --- a/api/walk.d.mts +++ b/api/walk.d.mts @@ -1,2 +1,2 @@ -import { a as asKeyValue, c as walk, i as WalkElement, n as EdgeTypeVariant, o as asSingle, r as Visitor, s as edgeLabel, t as EdgeType } from "./walk-Te23BaT5.mjs"; +import { a as asKeyValue, c as walk, i as WalkElement, n as EdgeTypeVariant, o as asSingle, r as Visitor, s as edgeLabel, t as EdgeType } from "./walk-B0aveFey.mjs"; export { EdgeType, EdgeTypeVariant, Visitor, WalkElement, asKeyValue, asSingle, edgeLabel, walk }; \ No newline at end of file diff --git a/eslint.config.mjs b/eslint.config.mjs index 6d76510..900f18d 100644 --- a/eslint.config.mjs +++ b/eslint.config.mjs @@ -4,7 +4,7 @@ import tsParser from "@typescript-eslint/parser"; import { resolve } from "node:path"; /* - * Strict, type-checked ESLint flat config for the @blockchaincommons/envelope library. + * Strict, type-checked ESLint flat config for the @blockchaincommons/dcbor library. */ const project = resolve(process.cwd(), "./tsconfig.json"); @@ -149,7 +149,7 @@ export default [ // Executable entry points inside a library: these run in Node.js and are // expected to use process, console and friends. { - files: ["src/bin/**/*.ts", "src/cmd/**/*.ts", "src/cli.ts", "src/main.ts"], + files: ["src/main.ts"], languageOptions: { parser: tsParser, parserOptions: { diff --git a/examples/comprehensive_walk_demo.ts b/examples/comprehensive_walk_demo.ts index d9b7537..968ca66 100644 --- a/examples/comprehensive_walk_demo.ts +++ b/examples/comprehensive_walk_demo.ts @@ -126,7 +126,7 @@ function main() { maxDepth: number; } - // P3: walk() returns void, so accumulate statistics in a closure-captured + // walk() returns void, so accumulate statistics in a closure-captured // object instead of threading them through the visitor state. const finalStats: Stats = { totalElements: 0, diff --git a/examples/debug_stop.ts b/examples/debug_stop.ts index 02bd3f8..116ff13 100644 --- a/examples/debug_stop.ts +++ b/examples/debug_stop.ts @@ -38,7 +38,7 @@ function main() { return [undefined, true]; } - // If we've seen the abort marker and this is an array at level 1, stop descent + // After the abort marker, stop descent into the array at index 2 const stop = foundAbort && element.type === "single" && diff --git a/examples/diagnostic_walk.ts b/examples/diagnostic_walk.ts index 3b60563..92e0fa2 100644 --- a/examples/diagnostic_walk.ts +++ b/examples/diagnostic_walk.ts @@ -1,7 +1,7 @@ /** * Diagnostic Walk Example * - * This example demonstrates using diagnostic_flat during tree traversal + * This example demonstrates flat diagnostic notation during tree traversal * to display CBOR elements at different nesting levels. * * Port of: bc-dcbor-rust/examples/diagnostic_walk.rs @@ -13,8 +13,8 @@ import { walk, type EdgeTypeVariant, type WalkElement } from "../src/walk"; import { diagnostic } from "../src/diag"; function main() { - // Test with various CBOR structures to show diagnostic_flat works during - // traversal + // Various CBOR structures, each printed with `diagnostic(…, { flat: true })` + // as a whole and then element by element during traversal. const testCases: Array<[string, any]> = [ ["Simple array", cbor([1, 2, 3])], [ diff --git a/examples/improved_text_collection.ts b/examples/improved_text_collection.ts index b16d394..33a7381 100644 --- a/examples/improved_text_collection.ts +++ b/examples/improved_text_collection.ts @@ -26,7 +26,8 @@ function main() { const texts: string[] = []; walk(cborData, undefined, (element: WalkElement, _depth, _edge, _state: void) => { - // Now we can collect ALL text nodes with a simple pattern match! + // Map keys and values are also visited as single elements, so one check + // collects every text node. if (element.type === "single") { if (element.cbor.type === MajorType.Text) { texts.push(element.cbor.value as string); diff --git a/examples/numeric_reduction.ts b/examples/numeric_reduction.ts new file mode 100644 index 0000000..f12839c --- /dev/null +++ b/examples/numeric_reduction.ts @@ -0,0 +1,68 @@ +/** + * Numeric Reduction Example + * + * Port of: bc-dcbor-rust/examples/numeric_reduction.rs + * + * - Integer extraction works across every integer width and float type that + * can hold the value. + * - A float with no fractional part (42.0) is reduced to an integer by dCBOR's + * canonical encoding rules, so it extracts as both a float and an integer. + * - A float with a fractional part (1.5) stays a float. It extracts as a float + * but not as an integer: `expectInteger` throws `WrongType`. + * - The last section reads the node's major type directly: unsigned, negative + * (which stores -1 - n) or a float simple value. 42.0 prints as unsigned(42) + * because numeric reduction already made it an integer at construction. + */ + +import { cbor, MajorType, type Cbor } from "../src/cbor"; +import { expectFloat, expectInteger, expectUnsigned } from "../src/conveniences-expect"; +import { isFloat } from "../src/conveniences-guards"; +import { CborError } from "../src/error"; + +function printNumeric(c: Cbor): void { + if (c.type === MajorType.Unsigned) { + console.log(`unsigned(${c.value})`); + } else if (c.type === MajorType.Negative) { + console.log(`negative(${-1n - BigInt(c.value)})`); + } else if (isFloat(c)) { + console.log(`float(${c.value.value})`); + } else { + throw new Error("not numeric"); + } +} + +function main() { + // Encode an integer and extract it at several widths. + const int = cbor(42); + const a = expectUnsigned(int, { width: 8 }); + const b = expectInteger(int); + const c = expectUnsigned(int, { width: 64 }); + const d = expectFloat(int); + console.log(`${a} ${b} ${c} ${d}`); + // 42 42 42 42 + + // A float with no fractional part: numeric reduction encodes 42.0 as the + // integer 42. + const whole = cbor(42.0); + console.log(`${expectFloat(whole)} ${expectInteger(whole)}`); + // 42 42 + + // A float with a fractional part stays a float. + const fraction = cbor(1.5); + let asInt: string; + try { + asInt = String(expectInteger(fraction)); + } catch (e) { + asInt = CborError.isCborError(e) ? `Err(${e.code})` : String(e); + } + console.log(`${expectFloat(fraction)} ${asInt}`); + // 1.5 Err(WrongType) + + // Interrogate the CBOR type directly. + printNumeric(cbor(42)); // unsigned(42) + printNumeric(cbor(-7)); // negative(-7) + printNumeric(cbor(42.0)); // unsigned(42) - numeric reduction + printNumeric(cbor(1.5)); // float(1.5) +} + +main(); diff --git a/examples/stop_behavior_test.ts b/examples/stop_behavior_test.ts index 87ad165..9074985 100644 --- a/examples/stop_behavior_test.ts +++ b/examples/stop_behavior_test.ts @@ -16,14 +16,15 @@ import { MajorType } from "../src/cbor"; function main() { console.log("=== Testing stop flag behavior ===\n"); - // Create a nested structure to test stop behavior + // Stopping only prevents descent into the stopped element's children, so + // the siblings of "stop_here" are still visited. const innerMap = new CborMap(); innerMap.set("inner_key", "inner_value"); const map = new CborMap(); - map.set("first", "stop_here"); // Will trigger stop - map.set("second", innerMap); // Should not be visited if stop works - map.set("third", [1, 2, 3]); // Should not be visited if stop works + map.set("first", "stop_here"); + map.set("second", innerMap); + map.set("third", [1, 2, 3]); const cborData = cbor(map); console.log(`CBOR structure: ${diagnostic(cborData, { flat: true })}`); diff --git a/examples/tagged_final_demo.ts b/examples/tagged_final_demo.ts index 8758283..d1a2d6a 100644 --- a/examples/tagged_final_demo.ts +++ b/examples/tagged_final_demo.ts @@ -60,7 +60,7 @@ function main() { console.log("\n=== Benefits of the new design ==="); console.log("✅ Tagged values are treated as semantic units"); - console.log("✅ Tag information is still accessible via cbor.tag()"); + console.log("✅ Tag information is still accessible via cbor.tag"); console.log("✅ Less noise in the output"); console.log("✅ Content is still traversed recursively if it has nested structure"); console.log("✅ Consistent with map key-value pair handling"); diff --git a/examples/text_collection_demo.ts b/examples/text_collection_demo.ts index 51e3346..014a6a0 100644 --- a/examples/text_collection_demo.ts +++ b/examples/text_collection_demo.ts @@ -16,7 +16,7 @@ import { MajorType } from "../src/cbor"; function main() { // Create a simple map with text keys and values const map = new CborMap(); - map.set("name", "Alice"); // Both text - should be easy to collect + map.set("name", "Alice"); // Text key, text value map.set("age", 30); // Text key, number value map.set("nested", [1, 2]); // Text key, array value const cborData = cbor(map); diff --git a/examples/walk_demo.ts b/examples/walk_demo.ts index 66cb997..0f100e9 100644 --- a/examples/walk_demo.ts +++ b/examples/walk_demo.ts @@ -62,8 +62,8 @@ function main() { keyValuePairs: number; } - // P3: walk() returns void, so accumulate counts in a closure-captured - // object instead of threading them through the visitor state. + // walk() returns void, so accumulate counts in a closure-captured object + // instead of threading them through the visitor state. const finalCount: Counter = { total: 0, maps: 0, @@ -84,7 +84,6 @@ function main() { if (element.type === "keyvalue") { finalCount.keyValuePairs += 1; } else { - // element.type === 'single' switch (element.cbor.type) { case MajorType.Map: finalCount.maps += 1; diff --git a/scripts/api-report.ts b/scripts/api-report.ts index d7387fa..f2937d0 100644 --- a/scripts/api-report.ts +++ b/scripts/api-report.ts @@ -1,5 +1,5 @@ /** - * Public API report via @microsoft/api-extractor (P1.2). + * Public API report via @microsoft/api-extractor. * * Usage: * bun scripts/api-report.ts --local # (re)generate api/dcbor.api.md @@ -7,9 +7,8 @@ * * api-extractor requires a `.d.ts` entry point; tsdown emits `.d.mts`, so a * transient copy is made inside dist/ first. The committed report - * (api/dcbor.api.md) is the reviewable record of the public surface - - * "API deliberately unstable, wire frozen" is enforced by making every - * surface change a visible diff here and in api/index.d.mts. + * (api/dcbor.api.md) is the reviewable record of the public surface: every + * surface change is a visible diff here and in api/index.d.mts. */ import { copyFileSync, existsSync, rmSync } from "node:fs"; diff --git a/scripts/api-snapshot.ts b/scripts/api-snapshot.ts index e3d8ff8..d097ad9 100644 --- a/scripts/api-snapshot.ts +++ b/scripts/api-snapshot.ts @@ -1,9 +1,9 @@ /** * Public-API snapshot. * - * Snapshots the built public type declarations of EVERY entry point - * (dist/.d.mts) to api/.d.mts so any change to the public - * surface is a reviewable diff (P3.18: index + diagnostic + walk + debug). + * Snapshots the built public type declarations of every entry point (index, + * diagnostic, walk, debug) from dist/.d.mts to api/.d.mts, so + * any change to the public surface is a reviewable diff. * * bun run api:snapshot # write/update the snapshots from the current build * bun run api:check # fail if any built .d.mts differs from its snapshot diff --git a/scripts/generate-vectors.ts b/scripts/generate-vectors.ts index 2bfbafb..0a21a44 100644 --- a/scripts/generate-vectors.ts +++ b/scripts/generate-vectors.ts @@ -28,17 +28,25 @@ import { dirname, join } from "node:path"; import { fileURLToPath } from "node:url"; import * as src from "../src/index.ts"; +import { diagnostic } from "../src/diag.ts"; +import { hexAnnotated } from "../src/dump.ts"; import { - redesignedAdapterFor, + currentAdapterFor, + decodeErrorMessage, decodeOutcome, encodeOutcome, hexToBytes, + materialize, + bytesToHex, } from "../tests/vectors/recipes.ts"; import { encodeCorpus } from "../tests/vectors/encode-corpus.ts"; import { decodeCorpus } from "../tests/vectors/decode-corpus.ts"; +import { formatCorpus, type FormatConfig } from "../tests/vectors/format-corpus.ts"; +import { dateCorpus } from "../tests/vectors/date-corpus.ts"; +import { uintCorpus } from "../tests/vectors/uint-corpus.ts"; const root = join(dirname(fileURLToPath(import.meta.url)), ".."); -const api = redesignedAdapterFor(src); +const api = currentAdapterFor(src); /** Above this many bytes, fixtures store a digest instead of full hex. */ const DIGEST_THRESHOLD_BYTES = 512; @@ -81,10 +89,10 @@ for (const entry of encodeCorpus) { expect = { ok: false, code: outcome.code }; } else { // Round-trip lock: encoder output must decode and re-encode - // byte-identically (dCBOR determinism) - EXCEPT the frozen bare-Float - // quirk where the float ladder emits 0xfa floats for whole values that - // the decoder's canonicality re-encode then rejects. Those are pinned - // with `decodeRejects` so the quirk itself is frozen. + // byte-identically (dCBOR determinism). An output the decoder rejects is + // pinned with `decodeRejects` instead of failing the run; no current + // fixture needs it, but the escape hatch stays so such a quirk is + // recorded deliberately rather than silently. const back = decodeOutcome(api, hexToBytes(outcome.hex)); let decodeRejects; if (!back.ok) { @@ -125,10 +133,11 @@ for (const entry of decodeCorpus) { problems.push(`decode ${entry.name}: expected accept, got ${actual.code}`); continue; } - if (actual.hex !== entry.hex) { - problems.push( - `decode ${entry.name}: accepted but re-encoded to ${actual.hex} (not byte-identical)`, - ); + // Accepts re-encode byte-identically unless the entry pins the bytes + // the decoded node re-encodes to (whole-valued float heads -> integers). + const want = entry.expect.hex ?? entry.hex; + if (actual.hex !== want) { + problems.push(`decode ${entry.name}: accepted but re-encoded to ${actual.hex}, expected ${want}`); continue; } } else { @@ -147,7 +156,108 @@ for (const entry of decodeCorpus) { continue; } } - decodeFixtures.push({ name: entry.name, hex: entry.hex, expect: entry.expect, note: entry.note }); + // Rejections also pin the error message: the reference's `Display` text + // is part of the contract (the Rust harness compares `e.to_string()`). + const expect = entry.expect.ok + ? entry.expect + : { ...entry.expect, message: decodeErrorMessage(api, hexToBytes(entry.hex)) }; + decodeFixtures.push({ name: entry.name, hex: entry.hex, expect, note: entry.note }); +} + +// --------------------------------------------------------------------------- +// Format fixtures - the five textual renderings under a chosen tags store. +// --------------------------------------------------------------------------- + +const storeFor = (config: FormatConfig): src.TagsStore => { + const store = new src.TagsStore(); + if (config === "standard") src.registerStandardTags(store); + if (config === "standard+bignum") src.registerStandardTags(store, { bignum: true }); + return store; +}; + +const formatFixtures = []; +for (const entry of formatCorpus) { + let value: src.Cbor; + if ("hex" in entry.input) { + value = src.decodeCbor(hexToBytes(entry.input.hex)); + } else { + try { + value = src.cbor(materialize(entry.input.recipe, api) as src.CborInput); + } catch (e) { + problems.push(`format ${entry.name}: recipe does not construct (${String(e)})`); + continue; + } + } + const store = storeFor(entry.config); + const tags = entry.config === "none" ? "none" : store; + formatFixtures.push({ + name: entry.name, + input: entry.input, + config: entry.config, + ...(entry.build === undefined ? {} : { build: entry.build }), + expect: { + hex: bytesToHex(src.encodeCbor(value)), + diagnostic: diagnostic(value, { tags }), + annotated: diagnostic(value, { annotate: true, tags }), + flat: diagnostic(value, { flat: true, tags }), + summary: diagnostic(value, { summarize: true, tags }), + hexAnnotated: hexAnnotated(value, { tagsStore: store }), + }, + ...(entry.note === undefined ? {} : { note: entry.note }), + }); +} + +// --------------------------------------------------------------------------- +// Date fixtures - CborDate decode and display. +// --------------------------------------------------------------------------- + +const errorOutcome = (e: unknown): { ok: false; code: string; message: string } => { + const code = api.errorCode(e); + if (code === undefined) throw e; + return { ok: false, code, message: (e as Error).message }; +}; + +const dateFixtures = []; +for (const entry of dateCorpus) { + let expect; + if (entry.kind === "decode") { + try { + const date = src.CborDate.fromTaggedCbor(src.decodeCbor(hexToBytes(entry.hex))); + expect = { ok: true, hex: bytesToHex(src.encodeCbor(date.toCbor())), display: date.toString() }; + } catch (e) { + expect = errorOutcome(e); + } + dateFixtures.push({ name: entry.name, kind: entry.kind, hex: entry.hex, expect, ...(entry.note === undefined ? {} : { note: entry.note }) }); + } else { + try { + const date = materialize(entry.recipe, api) as src.CborDate; + expect = { ok: true, hex: bytesToHex(src.encodeCbor(date.toCbor())), display: date.toString() }; + } catch (e) { + const { code } = errorOutcome(e); + expect = { ok: false, code }; + } + dateFixtures.push({ name: entry.name, kind: entry.kind, recipe: entry.recipe, expect, ...(entry.note === undefined ? {} : { note: entry.note }) }); + } +} + +// --------------------------------------------------------------------------- +// Unsigned-extraction fixtures - expectUnsigned(cbor, { width, wrapNegative }). +// --------------------------------------------------------------------------- + +const uintFixtures = []; +for (const entry of uintCorpus) { + let expect; + try { + const value = src.expectUnsigned(src.decodeCbor(hexToBytes(entry.hex)), { + width: entry.width, + wrapNegative: true, + }); + expect = { ok: true, value: String(value) }; + } catch (e) { + const { code } = errorOutcome(e); + expect = { ok: false, code }; + } + uintFixtures.push({ name: entry.name, hex: entry.hex, width: entry.width, expect, ...(entry.note === undefined ? {} : { note: entry.note }) }); } if (problems.length > 0) { @@ -170,6 +280,18 @@ writeFileSync( join(root, "tests/vectors/decode-vectors.json"), JSON.stringify({ ...meta(decodeFixtures.length), vectors: decodeFixtures }, null, 1) + "\n", ); +writeFileSync( + join(root, "tests/vectors/format-vectors.json"), + JSON.stringify({ ...meta(formatFixtures.length), vectors: formatFixtures }, null, 1) + "\n", +); +writeFileSync( + join(root, "tests/vectors/date-vectors.json"), + JSON.stringify({ ...meta(dateFixtures.length), vectors: dateFixtures }, null, 1) + "\n", +); +writeFileSync( + join(root, "tests/vectors/uint-vectors.json"), + JSON.stringify({ ...meta(uintFixtures.length), vectors: uintFixtures }, null, 1) + "\n", +); const throwing = encodeFixtures.filter((f) => !f.expect.ok).length; const digests = encodeFixtures.filter((f) => f.expect.ok && f.expect.sha256).length; @@ -179,4 +301,11 @@ console.log( console.log( `decode-vectors.json: ${decodeFixtures.length} vectors (${decodeFixtures.filter((f) => !f.expect.ok).length} rejections)`, ); +console.log(`format-vectors.json: ${formatFixtures.length} vectors`); +console.log( + `date-vectors.json: ${dateFixtures.length} vectors (${dateFixtures.filter((f) => !f.expect.ok).length} rejections)`, +); +console.log( + `uint-vectors.json: ${uintFixtures.length} vectors (${uintFixtures.filter((f) => !f.expect.ok).length} rejections)`, +); console.log(`source commit: ${sourceCommit}`); diff --git a/src/bignum.ts b/src/bignum.ts index f9858c9..5fa0e03 100644 --- a/src/bignum.ts +++ b/src/bignum.ts @@ -42,7 +42,7 @@ const TAG_3_NEGATIVE_BIGNUM = 3; * * @param bytes - The magnitude byte string to validate * @param isNegative - Whether this is for a negative bignum (tag 3) - * @throws CborError with type NonCanonicalNumeric on validation failure + * @throws {CborError} `NonCanonicalNumeric` on validation failure */ export function validateBignumMagnitude(bytes: Uint8Array, isNegative: boolean): void { if (isNegative) { @@ -122,7 +122,7 @@ export function bytesToBigint(bytes: Uint8Array): bigint { * * @param value - A non-negative bigint (must be >= 0n) * @returns CBOR tagged value - * @throws CborError with type OutOfRange if value is negative + * @throws {CborError} `OutOfRange` if value is negative */ export function biguintToCbor(value: bigint): Cbor { if (value < 0n) { @@ -170,8 +170,8 @@ export function bigintToCbor(value: bigint): Cbor { * * @param cbor - A CBOR value that should be a byte string * @returns Non-negative bigint - * @throws CborError with type WrongType if not a byte string - * @throws CborError with type NonCanonicalNumeric if encoding is non-canonical + * @throws {CborError} `WrongType` if not a byte string + * @throws {CborError} `NonCanonicalNumeric` if encoding is non-canonical */ export function biguintFromUntaggedCbor(cbor: Cbor): bigint { if (cbor.type !== MajorType.ByteString) { @@ -194,8 +194,8 @@ export function biguintFromUntaggedCbor(cbor: Cbor): bigint { * * @param cbor - A CBOR value that should be a byte string * @returns Negative bigint - * @throws CborError with type WrongType if not a byte string - * @throws CborError with type NonCanonicalNumeric if encoding is non-canonical + * @throws {CborError} `WrongType` if not a byte string + * @throws {CborError} `NonCanonicalNumeric` if encoding is non-canonical */ export function bigintFromNegativeUntaggedCbor(cbor: Cbor): bigint { if (cbor.type !== MajorType.ByteString) { diff --git a/src/byte-string.ts b/src/byte-string.ts index 3f20da1..d48255b 100644 --- a/src/byte-string.ts +++ b/src/byte-string.ts @@ -1,11 +1,5 @@ /** - * Byte string utilities for dCBOR. - * - * Represents a CBOR byte string (major type 2). - * - * `ByteString` is a wrapper around a byte array, optimized for use in CBOR - * encoding and decoding operations. It provides a richer API for working with - * byte data in the context of CBOR compared to using raw `Uint8Array` values. + * A mutable wrapper around a CBOR byte string (major type 2). * * In dCBOR, byte strings follow the general deterministic encoding rules: * - They must use definite-length encoding @@ -67,7 +61,7 @@ export class ByteString { * const bytes1 = new ByteString(new Uint8Array([1, 2, 3, 4])); * * // From a number array - * const bytes2 = new ByteString(new Uint8Array([5, 6, 7, 8])); + * const bytes2 = new ByteString([5, 6, 7, 8]); * ``` */ constructor(data: Uint8Array | number[]) { @@ -99,23 +93,15 @@ export class ByteString { } /** - * Returns a reference to the underlying byte data. - * - * @returns The raw bytes + * The underlying byte data. This is a live reference: mutations are + * visible to the `ByteString`; use `toUint8Array()` for a copy. * * @example * ```typescript * const bytes = new ByteString(new Uint8Array([1, 2, 3, 4])); - * assert.deepEqual(bytes.bytes, new Uint8Array([1, 2, 3, 4])); - * - * // You can use standard slice operations on the result * assert.deepEqual(bytes.bytes.slice(1, 3), new Uint8Array([2, 3])); * ``` */ - /** - * The underlying byte data (LIVE reference - mutations are visible to the - * ByteString; use `toUint8Array()` for a copy). - */ get bytes(): Uint8Array { return this._data; } @@ -209,7 +195,7 @@ export class ByteString { * * @param cbor - CBOR value * @returns ByteString if successful - * @throws Error if the CBOR value is not a byte string + * @throws {CborError} `WrongType` if the CBOR value is not a byte string * * @example * ```typescript @@ -222,7 +208,7 @@ export class ByteString { * try { * ByteString.fromCbor(cborInt); // throws * } catch(e) { - * // Error: Wrong type + * // CborError with code "WrongType" * } * ``` */ diff --git a/src/cbor-types.ts b/src/cbor-types.ts index 4a3af67..9010e07 100644 --- a/src/cbor-types.ts +++ b/src/cbor-types.ts @@ -34,11 +34,10 @@ export const MajorType = { export type MajorType = (typeof MajorType)[keyof typeof MajorType]; /** - * Numeric type that can be encoded in CBOR. - * - * Supports both standard JavaScript numbers and BigInt for large integers. - * Numbers are automatically encoded as either unsigned or negative integers - * depending on their value, following dCBOR canonical encoding rules. + * Numeric type that can be encoded in CBOR: a JavaScript `number` or a + * `bigint` for integers beyond the safe-integer range. Whole values encode as + * unsigned or negative integers; other numbers encode as floats, following + * dCBOR numeric reduction. * * @example * ```typescript @@ -49,16 +48,14 @@ export type MajorType = (typeof MajorType)[keyof typeof MajorType]; export type CborNumber = number | bigint; /** - * Type for values that can be converted to CBOR. - * - * This is a comprehensive union type representing all values that can be encoded - * as CBOR using the `cbor()` function. It includes: - * - Already-encoded CBOR values (`Cbor`) - * - Primitive types: numbers, bigints, strings, booleans, null, undefined - * - Binary data: `Uint8Array`, `ByteString` - * - Dates: `CborDate` - * - Collections: `CborMap`, arrays, JavaScript `Map`, JavaScript `Set` - * - Objects: Plain objects are converted to CBOR maps + * Values accepted by `cbor()`: + * - `Cbor` nodes + * - numbers, bigints, strings, booleans, null, undefined + * - `Uint8Array`, `ByteString` + * - `CborDate` + * - `CborMap`, arrays, JavaScript `Map` and `Set` + * - objects implementing `ToCbor` + * - plain objects, which become CBOR maps * * @example * ```typescript @@ -136,6 +133,14 @@ export interface CborTaggedType { readonly isCbor: true; readonly type: typeof MajorType.Tagged; readonly tag: CborNumber; + /** + * The name the tag was built with, when {@link taggedValue} received a + * named `Tag`. Never on the wire and never consulted by encoding, equality, + * `diagnostic` or `hexAnnotated` (those ask the tags store); it only + * surfaces in a `WrongTag` error's "but got …" text, as the reference's + * `Tag` carried through `CBORCase::Tagged` does. Decoded nodes have none. + */ + readonly tagName?: string | undefined; readonly value: Cbor; } export interface CborSimpleType { @@ -164,9 +169,9 @@ export interface CborMethods { toString(): string; } -// The `Cbor` union type is defined in ./cbor (where it merges with the `Cbor` -// namespace value); it is imported above as a type-only reference so the -// interfaces here can name it without a runtime dependency. +// The `Cbor` union type is defined in ./cbor; it is imported above as a +// type-only reference so the interfaces here can name it without a runtime +// dependency. /** * The structural encode protocol: types that can convert themselves to CBOR. diff --git a/src/cbor.ts b/src/cbor.ts index eb70928..a9d8e5f 100644 --- a/src/cbor.ts +++ b/src/cbor.ts @@ -18,12 +18,13 @@ */ import { CborMap } from "./map"; -import { simpleCborData } from "./simple"; +import { simpleCborData, simpleEquals } from "./simple"; import { hasFractionalPart } from "./float"; import { encodeVarInt, writeVarInt } from "./varint"; import { BufWriter } from "./buf-writer"; import { bytesToHex } from "./hex"; -import { type Tag } from "./tag"; +import { areBytesEqual } from "./stdlib"; +import { type Tag, tagValuesEqual } from "./tag"; import { CborError } from "./error"; import { U64_MAX, CBOR_INT_MIN, CBOR_INT_MAX } from "./numeric"; import { @@ -71,7 +72,7 @@ export type { * the (deliberately tiny) instance-method set from {@link CborMethods} * attached. The constituent interfaces live in ./cbor-types. * - * This is a TYPE-ONLY export. Construct values with `cbor(x)` and decode with + * A type-only export: construct values with `cbor(x)` and decode with * `decodeCbor(bytes)`. */ export type Cbor = ( @@ -183,21 +184,65 @@ const CBOR_NULL = attachMethods({ // ============================================================================ /** - * Structural CBOR value equality. + * Structural CBOR value equality, the reference's `PartialEq for CBOR` + * (`cbor.rs`): two values are equal when they have the same major type and + * equal contents, compared recursively. * - * dCBOR encoding is deterministic, so two CBOR values are equal iff they - * encode to the same byte sequence - this is the simplest correct comparator. + * This is not "encode to the same bytes": a float node whose value is whole + * (`Float(2.0)`, reachable through a bare node) encodes as the integer `2` + * but is not equal to the integer node; a text node keeps the string it was + * built from, so a decomposed `"é"` is not equal to the composed one although + * both encode composed. Integers compare by value across `number`/`bigint`; + * tags compare by value only (the carried name is ignored); NaN equals NaN; + * maps compare entry by entry in canonical key order, keys and values both + * structurally. * * Use this rather than `===` (which compares JS object references) when * you need value equality across two `Cbor` instances built independently. */ export const cborEquals = (a: Cbor, b: Cbor): boolean => { if (a === b) return true; - const aBytes = encodeCbor(a); - const bBytes = encodeCbor(b); - if (aBytes.length !== bBytes.length) return false; - for (let i = 0; i < aBytes.length; i++) { - if (aBytes[i] !== bBytes[i]) return false; + switch (a.type) { + case MajorType.Unsigned: + return b.type === MajorType.Unsigned && BigInt(a.value) === BigInt(b.value); + case MajorType.Negative: + return b.type === MajorType.Negative && BigInt(a.value) === BigInt(b.value); + case MajorType.ByteString: + return b.type === MajorType.ByteString && areBytesEqual(a.value, b.value); + case MajorType.Text: + return b.type === MajorType.Text && a.value === b.value; + case MajorType.Array: + return ( + b.type === MajorType.Array && + a.value.length === b.value.length && + a.value.every((item, i) => cborEquals(item, b.value[i])) + ); + case MajorType.Map: + return b.type === MajorType.Map && mapEquals(a.value, b.value); + case MajorType.Tagged: + return ( + b.type === MajorType.Tagged && tagValuesEqual(a.tag, b.tag) && cborEquals(a.value, b.value) + ); + case MajorType.Simple: + return b.type === MajorType.Simple && simpleEquals(a.value, b.value); + } +}; + +/** + * `PartialEq for Map` (`map.rs`): the same entries in canonical key order, + * each with a structurally equal stored key node and value node. Both maps + * iterate in encoded-key order, so a lockstep walk is exact, and like the + * reference's `BTreeMap` equality it stops at the first mismatch. (The + * reference also compares the stored key bytes; that is implied here, since + * structurally equal key nodes always encode to the same bytes.) + */ +const mapEquals = (a: CborMap, b: CborMap): boolean => { + const n = a.size; + if (n !== b.size) return false; + for (let i = 0; i < n; i++) { + const l = a.entryAt(i); + const r = b.entryAt(i); + if (!cborEquals(l.key, r.key) || !cborEquals(l.value, r.value)) return false; } return true; }; @@ -235,7 +280,7 @@ const hasToCbor = (value: unknown): value is ToCbor => { * @example * ```typescript * cbor(42); // integer - * cbor("héllo"); // NFC-normalized text + * cbor("héllo"); // text (NFC-normalized when encoded) * cbor([1, "two", true, null]); // array * cbor(new Map([["k", 1]])); // map (canonical key order) * cbor({ name: "Alice", age: 30 }); // plain object -> map @@ -309,10 +354,10 @@ export const cbor = (value: CborInput): Cbor => { result = { isCbor: true, type: MajorType.Unsigned, value: value }; } } else if (typeof value === "string") { - // dCBOR requires all text strings to be in Unicode Normalization Form C (NFC) - // This ensures deterministic encoding regardless of how the string was composed - const normalized = value.normalize("NFC"); - result = { isCbor: true, type: MajorType.Text, value: normalized }; + // The node keeps the string exactly as given (the reference's + // `CBORCase::Text(String)` does the same); dCBOR's NFC requirement is + // applied when the node is encoded, see `writeCborInto`. + result = { isCbor: true, type: MajorType.Text, value }; } else if (value === null || value === undefined) { return CBOR_NULL; } else if (value === true) { @@ -341,7 +386,7 @@ export const cbor = (value: CborInput): Cbor => { // Directive error: objects implementing taggedCbor() are not auto-wrapped. // Auto-wrapping would change bytes for structural call sites, so throw. throw CborError.custom( - "objects implementing taggedCbor() are no longer auto-wrapped by cbor(); " + + "objects implementing taggedCbor() are not auto-wrapped by cbor(); " + "implement toCbor() (e.g. `toCbor() { return this.taggedCbor(); }`)", ); } else if (typeof value === "object" && "tag" in value && "value" in value) { @@ -351,7 +396,7 @@ export const cbor = (value: CborInput): Cbor => { // it could be a tagged value or a legitimate data record like // {tag: "release", value: 3}. Refuse it rather than guess. throw CborError.custom( - "plain { tag, value } objects are ambiguous and no longer encode as tagged values; " + + "plain { tag, value } objects are ambiguous and do not encode as tagged values; " + "use taggedValue(tag, content) for a tagged value, or add/rename a key to encode a map", ); } @@ -384,6 +429,22 @@ export const cbor = (value: CborInput): Cbor => { // avoids a per-string allocation on the encode path. const textEncoder = new TextEncoder(); +/** + * dCBOR requires every encoded text string to be in Unicode Normalization + * Form C. Like the reference (`cbor.rs`: `x.nfc().collect()` inside + * `cbor_data`), normalization happens here at encode time, so the node keeps + * the string it was built from. Strings whose code units are all below U+0300 + * (the first combining mark) contain nothing that can compose or decompose + * and are already NFC; skipping `normalize` for them keeps the ASCII/Latin-1 + * hot path allocation-free. + */ +const toNfc = (text: string): string => { + for (let i = 0; i < text.length; i++) { + if (text.charCodeAt(i) >= 0x300) return text.normalize("NFC"); + } + return text; +}; + /** * Write a CBOR value into `writer`. The whole tree encodes into one growable * buffer, so nested containers don't allocate-and-concatenate a fresh array @@ -408,7 +469,7 @@ const writeCborInto = (writer: BufWriter, value: CborInput): void => { break; case MajorType.Text: if (typeof c.value === "string") { - const utf8Bytes = textEncoder.encode(c.value); + const utf8Bytes = textEncoder.encode(toNfc(c.value)); writeVarInt(writer, utf8Bytes.length, MajorType.Text); writer.writeBytes(utf8Bytes); return; @@ -431,11 +492,17 @@ const writeCborInto = (writer: BufWriter, value: CborInput): void => { } return; case MajorType.Map: { - const entries = c.value.entriesArray; - writeVarInt(writer, entries.length, MajorType.Map); - for (const { key, value: entryValue } of entries) { - writeCborInto(writer, key); - writeCborInto(writer, entryValue); + const map = c.value; + const n = map.size; + writeVarInt(writer, n, MajorType.Map); + for (let i = 0; i < n; i++) { + // Write the stored encoded key bytes - the bytes the entry is sorted + // by - as the reference's `Map::cbor_data` (`map.rs`) writes its + // `MapKey`. Re-encoding the key node would produce the same bytes + // (encoding is deterministic) at the cost of a second NFC + UTF-8 + // pass for every text key. + writer.writeBytes(map.encodedKeyAt(i)); + writeCborInto(writer, map.entryAt(i).value); } return; } @@ -483,7 +550,7 @@ export const encodeCbor = (value: CborInput): Uint8Array => { // ============================================================================ /** - * Construct a tagged value - the ONLY explicit tagged-value constructor. + * Construct a tagged value - the only explicit tagged-value constructor. * * @example * ```typescript @@ -491,17 +558,30 @@ export const encodeCbor = (value: CborInput): Uint8Array => { * taggedValue(Tag.from(32), "https://example.com/"); // URI, tag 32 * ``` * - * @param tag - The tag number (`number | bigint`) or a `Tag` object (its - * `.value` is used; names never reach the wire). + * @param tag - The tag number (`number | bigint`) or a `Tag` object. Its + * `.value` goes on the wire; a `.name` is kept on the node (see + * `CborTaggedType.tagName`) so a `WrongTag` error can name the tag that + * was found, as the reference does. * @param content - Anything `cbor()` accepts. * @public */ export const taggedValue = (tag: CborNumber | Tag, content: CborInput): Cbor => { - const tagVal = typeof tag === "object" && "value" in tag ? tag.value : tag; - return attachMethods({ - isCbor: true, - type: MajorType.Tagged, - tag: tagVal, - value: cbor(content), - }); + if (typeof tag === "object" && "value" in tag) { + if (tag.name !== undefined) { + return attachMethods({ + isCbor: true, + type: MajorType.Tagged, + tag: tag.value, + tagName: tag.name, + value: cbor(content), + }); + } + return attachMethods({ + isCbor: true, + type: MajorType.Tagged, + tag: tag.value, + value: cbor(content), + }); + } + return attachMethods({ isCbor: true, type: MajorType.Tagged, tag, value: cbor(content) }); }; diff --git a/src/codable.ts b/src/codable.ts index 04640dc..01ddc19 100644 --- a/src/codable.ts +++ b/src/codable.ts @@ -15,7 +15,7 @@ import { type Cbor } from "./cbor"; import { MajorType } from "./cbor-types"; -import { tagValuesEqual, type Tag } from "./tag"; +import { Tag, tagValuesEqual } from "./tag"; import { CborError } from "./error"; import { decodeCbor } from "./decode"; @@ -37,7 +37,7 @@ export interface CborTagged { * The codec value is the runtime witness that justifies the generic in * {@link decodeWith} - no unwitnessed casts. * - * Ship-with exemplar: `CborDate.codec`. + * Example implementation: `CborDate.codec`. * * @beta */ @@ -70,7 +70,7 @@ export function decodeWith(data: Uint8Array, codec: CborCodec): T { } /** - * Helper function to validate that a CBOR value has one of the expected tags. + * Validate that a CBOR value has one of the expected tags. * * @param cbor - CBOR value to validate * @param expectedTags - Array of valid tags @@ -86,16 +86,17 @@ export const validateTag = (cbor: Cbor, expectedTags: Tag[]): Tag => { const tagValue = cbor.tag; const matchingTag = expectedTags.find((t) => tagValuesEqual(t.value, tagValue)); if (matchingTag === undefined) { - // Produce the structured WrongTag variant rather than a stringly-typed - // Custom error so callers can branch on `error.code === "WrongTag"`. - throw CborError.wrongTag(expectedTags[0], { value: tagValue }); + // Both tags keep their names, as the reference's `WrongTag(Tag, Tag)` + // does: the expected one as `cborTags()` returned it, the actual one as + // the node carries it (a decoded node carries no name). + throw CborError.wrongTag(expectedTags[0], Tag.from(tagValue, cbor.tagName)); } return matchingTag; }; /** - * Helper function to extract the content from a tagged CBOR value. + * Extract the content from a tagged CBOR value. * * @param cbor - Tagged CBOR value * @returns The untagged content diff --git a/src/conveniences-accessors.ts b/src/conveniences-accessors.ts index 0be395e..9469dee 100644 --- a/src/conveniences-accessors.ts +++ b/src/conveniences-accessors.ts @@ -69,9 +69,7 @@ export const asInteger = (cbor: Cbor): number | bigint | undefined => { * * Decoded byte strings are zero-copy views aliasing the input buffer - * mutating the input after decoding (or mutating the returned bytes) changes - * the other side. Call `.slice()` first if you need an independent copy. This - * is deliberate: the zero-copy decode performance profile is part of the - * library's contract. + * the other side. Call `.slice()` first if you need an independent copy. * * @param cbor - CBOR value * @returns Byte string or undefined @@ -232,10 +230,9 @@ export const arrayLength = (cbor: Cbor): number | undefined => { * Check if array is empty. * * @param cbor - CBOR value (must be array) - * @returns True if empty, false if not empty, undefined if not array + * @returns True if empty; false if not empty or not an array (matching hasTag) */ export const arrayIsEmpty = (cbor: Cbor): boolean => { - // A wrong major type returns plain `false` (matching hasTag). if (cbor.type !== MajorType.Array) { return false; } @@ -266,10 +263,9 @@ export function mapValue(cbor: Cbor, key: CborInput): Cbor | undefined { * * @param cbor - CBOR value (must be map) * @param key - Map key - * @returns True if key exists, false otherwise, undefined if not map + * @returns True if key exists; false otherwise or if not a map (matching hasTag) */ export function mapHas(cbor: Cbor, key: CborInput): boolean { - // A wrong major type returns plain `false` (matching hasTag). if (cbor.type !== MajorType.Map) { return false; } @@ -319,10 +315,9 @@ export const mapSize = (cbor: Cbor): number | undefined => { * Check if map is empty. * * @param cbor - CBOR value (must be map) - * @returns True if empty, false if not empty, undefined if not map + * @returns True if empty; false if not empty or not a map (matching hasTag) */ export const mapIsEmpty = (cbor: Cbor): boolean => { - // A wrong major type returns plain `false` (matching hasTag). if (cbor.type !== MajorType.Map) { return false; } @@ -401,8 +396,7 @@ export const asTaggedValue = (cbor: Cbor): [Tag, Cbor] | undefined => { if (cbor.type !== MajorType.Tagged) { return undefined; } - // Resolve the canonical name (if any) via the global tags store rather - // than synthesizing a `tag-${value}` placeholder. + // The tag carries the global tags store's name for it, if any. const resolved = getGlobalTagsStore().tagForValue(cbor.tag); const tag: Tag = resolved ?? { value: cbor.tag }; return [tag, cbor.value]; diff --git a/src/conveniences-expect.ts b/src/conveniences-expect.ts index 16a7eaa..51652a8 100644 --- a/src/conveniences-expect.ts +++ b/src/conveniences-expect.ts @@ -9,8 +9,9 @@ import { type Cbor } from "./cbor"; import { MajorType, type CborNumber } from "./cbor-types"; import type { CborMap } from "./map"; import { isFloat as isSimpleFloat } from "./simple"; -import { tagValuesEqual } from "./tag"; +import { Tag, tagValuesEqual } from "./tag"; import { CborError } from "./error"; +import { narrowInteger } from "./numeric"; import { asUnsigned, asNegative, @@ -24,19 +25,62 @@ import { asNumber, } from "./conveniences-accessors"; +/** + * Options for {@link expectUnsigned}: extract into a fixed-width unsigned + * integer the way the reference's `u8`/`u16`/`u32`/`u64: TryFrom` + * do (dcbor 0.25.2 `int.rs`). + */ +export interface ExpectUnsignedOptions { + /** Target width in bits; an unsigned value above 2^width − 1 is `OutOfRange`. */ + readonly width: 8 | 16 | 32 | 64; + /** + * Also accept a negative integer node and wrap it as the reference does: + * a value `v` in [−2^width, −1] yields `2^width + v` (so −1 is 255 at + * width 8), and a value below −2^width is `OutOfRange`. Off by default: + * without it a negative node is `WrongType`, as for every other type. See + * RUST_DIVERGENCES.md §1.1. + */ + readonly wrapNegative?: boolean | undefined; +} + /** * Extract unsigned integer value, throwing if type doesn't match. * + * With `options`, the value is checked against a fixed width and, when + * `wrapNegative` is set, a negative node is wrapped exactly as the + * reference's `u*::try_from` wraps it (see {@link ExpectUnsignedOptions}). + * * @param cbor - CBOR value - * @returns Unsigned integer - * @throws {CborError} With type 'WrongType' if cbor is not an unsigned integer + * @param options - Fixed-width extraction (optional; without it the + * behaviour is the plain `Unsigned`-or-`WrongType` check) + * @returns Unsigned integer (`bigint` above `Number.MAX_SAFE_INTEGER`) + * @throws {CborError} `WrongType` if cbor is not an unsigned integer (or, + * with `wrapNegative`, not an integer); `OutOfRange` when the value does + * not fit `width` */ -export const expectUnsigned = (cbor: Cbor): number | bigint => { - const value = asUnsigned(cbor); - if (value === undefined) { - throw CborError.wrongType(); +export const expectUnsigned = (cbor: Cbor, options?: ExpectUnsignedOptions): number | bigint => { + if (options === undefined) { + const value = asUnsigned(cbor); + if (value === undefined) { + throw CborError.wrongType(); + } + return value; } - return value; + // `From64::from_u64(n, MAX)`: the magnitude on the wire must fit the width. + const max = (1n << BigInt(options.width)) - 1n; + if (cbor.type === MajorType.Unsigned) { + const value = BigInt(cbor.value); + if (value > max) throw CborError.outOfRange(); + return narrowInteger(value); + } + if (cbor.type === MajorType.Negative && options.wrapNegative === true) { + // The node stores the magnitude m = −1 − v; the reference computes + // `(-1 - m) as uN`, i.e. 2^width + v = max − m, after the same range check. + const magnitude = BigInt(cbor.value); + if (magnitude > max) throw CborError.outOfRange(); + return narrowInteger(max - magnitude); + } + throw CborError.wrongType(); }; /** @@ -44,7 +88,7 @@ export const expectUnsigned = (cbor: Cbor): number | bigint => { * * @param cbor - CBOR value * @returns Negative integer - * @throws {CborError} With type 'WrongType' if cbor is not a negative integer + * @throws {CborError} `WrongType` if cbor is not a negative integer */ export const expectNegative = (cbor: Cbor): number | bigint => { const value = asNegative(cbor); @@ -59,7 +103,7 @@ export const expectNegative = (cbor: Cbor): number | bigint => { * * @param cbor - CBOR value * @returns Integer - * @throws {CborError} With type 'WrongType' if cbor is not an integer + * @throws {CborError} `WrongType` if cbor is not an integer */ export const expectInteger = (cbor: Cbor): number | bigint => { const value = asInteger(cbor); @@ -72,15 +116,13 @@ export const expectInteger = (cbor: Cbor): number | bigint => { /** * Extract byte string value, throwing if type doesn't match. * + * Decoded byte strings are zero-copy views aliasing the input buffer - + * mutating the input after decoding (or mutating the returned bytes) changes + * the other side. Call `.slice()` first if you need an independent copy. + * * @param cbor - CBOR value * @returns Byte string - * @throws {CborError} With type 'WrongType' if cbor is not a byte string - * - * NOTE: decoded byte strings are zero-copy views aliasing the input - * buffer - mutating the input after decoding (or mutating the returned - * bytes) changes the other side. Call `.slice()` first if you need an - * independent copy. This is deliberate: the zero-copy decode performance - * profile is part of the library's contract. + * @throws {CborError} `WrongType` if cbor is not a byte string */ export const expectBytes = (cbor: Cbor): Uint8Array => { const value = asBytes(cbor); @@ -95,7 +137,7 @@ export const expectBytes = (cbor: Cbor): Uint8Array => { * * @param cbor - CBOR value * @returns Text string - * @throws {CborError} With type 'WrongType' if cbor is not a text string + * @throws {CborError} `WrongType` if cbor is not a text string */ export const expectText = (cbor: Cbor): string => { const value = asText(cbor); @@ -110,7 +152,7 @@ export const expectText = (cbor: Cbor): string => { * * @param cbor - CBOR value * @returns Array - * @throws {CborError} With type 'WrongType' if cbor is not an array + * @throws {CborError} `WrongType` if cbor is not an array */ export const expectArray = (cbor: Cbor): readonly Cbor[] => { const value = asArray(cbor); @@ -125,7 +167,7 @@ export const expectArray = (cbor: Cbor): readonly Cbor[] => { * * @param cbor - CBOR value * @returns Map - * @throws {CborError} With type 'WrongType' if cbor is not a map + * @throws {CborError} `WrongType` if cbor is not a map */ export const expectMap = (cbor: Cbor): CborMap => { const value = asMap(cbor); @@ -140,7 +182,7 @@ export const expectMap = (cbor: Cbor): CborMap => { * * @param cbor - CBOR value * @returns Boolean - * @throws {CborError} With type 'WrongType' if cbor is not a boolean + * @throws {CborError} `WrongType` if cbor is not a boolean */ export const expectBoolean = (cbor: Cbor): boolean => { const value = asBoolean(cbor); @@ -153,13 +195,14 @@ export const expectBoolean = (cbor: Cbor): boolean => { /** * Extract float value, throwing if type doesn't match. * + * Integers coerce to float, as for {@link asFloat}. + * * @param cbor - CBOR value * @returns Float - * @throws {CborError} With type 'WrongType' if cbor is not a float + * @throws {CborError} `WrongType` if cbor is not numeric; `OutOfRange` if an + * integer is not exactly representable as f64 */ export const expectFloat = (cbor: Cbor): number => { - // Numeric types coerce to float (OutOfRange if an integer isn't exactly - // representable as f64); anything else is WrongType. if (cbor.type === MajorType.Unsigned || cbor.type === MajorType.Negative) { const value = asFloat(cbor); if (value === undefined) { @@ -178,7 +221,7 @@ export const expectFloat = (cbor: Cbor): number => { * * @param cbor - CBOR value * @returns Number - * @throws {CborError} With type 'WrongType' if cbor is not a number + * @throws {CborError} `WrongType` if cbor is not a number */ export const expectNumber = (cbor: Cbor): CborNumber => { const value = asNumber(cbor); @@ -189,21 +232,26 @@ export const expectNumber = (cbor: Cbor): CborNumber => { }; /** - * Extract content if has specific tag, throwing if not. + * Extract content if has specific tag, throwing if not (the reference's + * `try_into_expected_tagged_value`). * - * Throws `{ type: "WrongType" }` if `cbor` is not tagged at all, otherwise - * `{ type: "WrongTag", expected, actual }` if the tag doesn't match. + * The `WrongTag` error names the expected tag as it was given (a `Tag` keeps + * its name; a number or bigint stays unnamed) and the actual tag as the node + * carries it. * * @param cbor - CBOR value - * @param tag - Expected tag value + * @param tag - Expected tag value, or a `Tag` * @returns Tagged content + * @throws {CborError} `WrongType` if `cbor` is not tagged; `WrongTag` (with + * `details.expectedTag` and `details.actualTag`) if the tag doesn't match */ -export const expectTaggedContent = (cbor: Cbor, tag: number | bigint): Cbor => { +export const expectTaggedContent = (cbor: Cbor, tag: number | bigint | Tag): Cbor => { if (cbor.type !== MajorType.Tagged) { throw CborError.wrongType(); } - if (!tagValuesEqual(cbor.tag, tag)) { - throw CborError.wrongTag({ value: tag }, { value: cbor.tag }); + const expected = typeof tag === "object" ? tag : Tag.from(tag); + if (!tagValuesEqual(cbor.tag, expected.value)) { + throw CborError.wrongTag(expected, Tag.from(cbor.tag, cbor.tagName)); } return cbor.value; }; diff --git a/src/date.ts b/src/date.ts index 5f86d93..1f279cf 100644 --- a/src/date.ts +++ b/src/date.ts @@ -1,10 +1,9 @@ /** - * Date/time support for CBOR with tag(1) encoding. + * Date/time support for CBOR with tag 1 encoding. * - * A CBOR-friendly representation of a date and time. - * - * The `CborDate` type provides a wrapper around JavaScript's native `Date` that - * supports encoding and decoding to/from CBOR with tag 1, following the CBOR + * The `CborDate` type holds an instant as whole seconds since the Unix epoch + * plus nanoseconds - the same model as the reference's `chrono::DateTime` - + * and encodes and decodes it to/from CBOR with tag 1, following the CBOR * date/time standard specified in RFC 8949. * * When encoded to CBOR, dates are represented as tag 1 followed by a numeric @@ -19,26 +18,17 @@ import { type Cbor } from "./cbor"; import { MajorType } from "./cbor-types"; import { cbor, taggedValue } from "./cbor"; import { Tag } from "./tag"; -import { TAG_EPOCH_DATE_TIME } from "./tags"; +import { TAG_DATE } from "./tags"; +import { getGlobalTagsStore } from "./tags-store"; import { type CborTagged, type CborCodec, validateTag, extractTaggedContent } from "./codable"; import { CborError } from "./error"; -/** - * Normalize a timestamp (seconds since the Unix epoch) to whole seconds plus a - * non-negative, sub-second nanosecond part, so dates round-trip byte-identically. - * - * The nanosecond part is truncated toward zero and clamped to [0, u32::MAX]. So - * a negative fraction floors the value (`-1.5` becomes `-1.0`) and sub-nanosecond - * precision is dropped (`1.0000000005` becomes `1.0`). - * - * @internal - */ /** * The reference's representable range: chrono's `NaiveDateTime::MIN` * (−262143-01-01T00:00:00) and `MAX` (262142-12-31T23:59:59.999999999) as * Unix seconds. Beyond it `Date::from_timestamp` panics (`timestamp_opt(…) * .unwrap()`); here it is `InvalidDate`. JS `Date` reaches further (±8.64e12 - * s), so every accepted value also renders. + * s), so `toDate()` can represent every accepted value. */ const MIN_TIMESTAMP_SECONDS = -8_334_601_228_800; const MAX_TIMESTAMP_SECONDS = 8_210_266_876_799; @@ -96,7 +86,7 @@ function civilFromDays(days: number): [number, number, number] { * Whole seconds since the Unix epoch of the given UTC components, or * `undefined` when they are not a valid date-time. The checks are chrono's * (`NaiveDate::from_ymd_opt`, `NaiveTime::from_hms_opt`): the year within - * ±262143, a calendar-valid month and day, and `hh:mm:ss` within 23:59:59. + * −262143…+262142, a calendar-valid month and day, and `hh:mm:ss` within 23:59:59. */ function civilSeconds( year: number, @@ -138,47 +128,51 @@ const YMD = new RegExp( `^${WHITESPACE}*(?:([+-])(\\d+)|(\\d{1,4}))-${WHITESPACE}*(\\d{1,2})-${WHITESPACE}*(\\d{1,2})$`, ); -function normalizeTimestampSeconds(seconds: number): number { - if (!Number.isFinite(seconds)) { - // There is no representation for a non-finite instant; reject with a - // typed error (the reference saturates NaN to the epoch and panics on ±∞). - throw CborError.invalidDate("non-finite timestamp"); - } - if (seconds < MIN_TIMESTAMP_SECONDS || seconds >= MAX_TIMESTAMP_SECONDS + 1) { +/** + * Split a timestamp (seconds since the Unix epoch) into the (whole seconds, + * nanoseconds) pair the reference's `Date::from_timestamp` builds: + * + * - `trunc() as i64` for the seconds - NaN saturates to 0 (the epoch); + * ±Infinity saturates to the `i64` bounds, which chrono rejects and the + * reference then panics on, so here it is `InvalidDate`; + * - `(fract() * 1e9) as u32` for the nanoseconds - truncated toward zero and + * saturated to `[0, u32::MAX]`, so a negative fraction is dropped (`-1.5` + * becomes `-1`) and sub-nanosecond precision is lost; + * - `timestamp_opt(...)` then requires the whole seconds inside chrono's + * range (the fraction does not take part, so `MIN - 0.5` is `MIN`). + * + * @internal + */ +function timestampParts(seconds: number): [whole: number, nanoseconds: number] { + if (Number.isNaN(seconds)) return [0, 0]; + if (!Number.isFinite(seconds)) throw CborError.invalidDate("non-finite timestamp"); + const whole = Math.trunc(seconds); + if (whole < MIN_TIMESTAMP_SECONDS || whole > MAX_TIMESTAMP_SECONDS) { throw CborError.invalidDate("timestamp outside the representable range"); } - const whole = Math.trunc(seconds); - let nsecs = Math.trunc((seconds - whole) * 1_000_000_000); - if (nsecs < 0) { - nsecs = 0; - } else if (nsecs > 0xffffffff) { - nsecs = 0xffffffff; + let nanoseconds = Math.trunc((seconds - whole) * 1_000_000_000); + if (nanoseconds < 0) { + nanoseconds = 0; + } else if (nanoseconds > 0xffffffff) { + nanoseconds = 0xffffffff; } - return whole + nsecs / 1_000_000_000; + return [whole, nanoseconds]; } +let dateCodec: CborCodec | undefined; + /** - * A CBOR-friendly representation of a date and time. - * - * The `CborDate` type provides a wrapper around JavaScript's native `Date` that - * supports encoding and decoding to/from CBOR with tag 1, following the CBOR - * date/time standard specified in RFC 8949. + * A UTC date and time, encoded as CBOR tag 1 (RFC 8949 epoch-based + * date/time). * - * When encoded to CBOR, dates are represented as tag 1 followed by a numeric - * value representing the number of seconds since (or before) the Unix epoch - * (1970-01-01T00:00:00Z). The numeric value can be a positive or negative - * integer, or a floating-point value for dates with fractional seconds. - * - * # Features - * - * - Supports UTC dates with optional fractional seconds - * - Provides convenient constructors for common date creation patterns - * - Implements the `CborTagged` interface and the `ToCbor` protocol - * - Supports arithmetic operations with durations and between dates + * The instant is held as whole seconds since the Unix epoch plus nanoseconds. + * On the wire it is tag 1 followed by the seconds since (or before) + * 1970-01-01T00:00:00Z: an integer for whole seconds, a float otherwise. + * Implements the `CborTagged` interface and the `ToCbor` protocol. * * @example * ```typescript - * import { CborDate } from './date'; + * import { CborDate } from "@blockchaincommons/dcbor"; * * // Create a date from a timestamp (seconds since Unix epoch) * const date = CborDate.fromEpochSeconds(1675854714.0); @@ -193,8 +187,6 @@ function normalizeTimestampSeconds(seconds: number): number { * const decoded = CborDate.fromTaggedCbor(cborValue); * ``` */ -let dateCodec: CborCodec | undefined; - export class CborDate implements CborTagged { /** Debug label: `Object.prototype.toString` reports `[object CborDate]`. */ // A prototype getter has zero per-instance cost; the readonly field the @@ -205,30 +197,25 @@ export class CborDate implements CborTagged { } /** - * Canonical timestamp in seconds since the Unix epoch as a JS `number` - * (`f64`). dCBOR encodes Date (tag 1) as a numeric value in seconds, so - * keeping `_seconds` as the source of truth avoids the millisecond-only - * round-trip precision loss that going through a JS `Date` instance would - * introduce. - * - * f64 bounds the achievable precision (~16 decimal digits, so roughly - * microseconds for current epoch values), but the encode/decode round-trip - * is byte-identical. + * The instant as the reference's `chrono::DateTime` holds it: whole + * seconds since the Unix epoch plus a nanosecond part in + * `[0, 1_999_999_999]` (values from 10⁹ up represent a leap second, e.g. + * `23:59:60`, as chrono does). Keeping the pair rather than one `f64` + * means display, equality and ordering see exactly what the reference + * sees; the wire value is derived from it as `timestamp()` does. */ - private _seconds: number; + private readonly _seconds: number; + private readonly _nanoseconds: number; /** * Creates a new `CborDate` from the given JavaScript `Date`. * - * This method creates a new `CborDate` instance by wrapping a - * JavaScript `Date`. - * - * @param dateTime - A `Date` instance to wrap + * @param dateTime - A `Date` instance * * @returns A new `CborDate` instance * * @throws `InvalidDate` for an invalid `Date` (`NaN` time) or one outside - * the reference's representable range (±262143 years), which a chrono + * the reference's representable range (years −262143 to 262142), which a chrono * value handed to `Date::from_datetime` can never be. * * @example @@ -248,17 +235,13 @@ export class CborDate implements CborTagged { if (whole < MIN_TIMESTAMP_SECONDS || whole > MAX_TIMESTAMP_SECONDS) { throw CborError.invalidDate("timestamp outside the representable range"); } - const instance = new CborDate(); - // `timestamp()`: whole seconds plus nanoseconds over 10⁹ (the - // millisecond part is exact in nanoseconds). - instance._seconds = whole + ((ms - whole * 1000) * 1_000_000) / 1_000_000_000; - return instance; + // The millisecond part is exact in nanoseconds. + return new CborDate(whole, (ms - whole * 1000) * 1_000_000); } /** - * Creates a new `CborDate` from year, month, and day components. - * - * This method creates a new `CborDate` with the time set to 00:00:00 UTC. + * Creates a new `CborDate` from year, month, and day components, at + * 00:00:00 UTC. * * @param year - The year component (e.g., 2023) * @param month - The month component (1-12) @@ -300,7 +283,7 @@ export class CborDate implements CborTagged { * * @throws `InvalidDate` if the components do not form a valid date and time * — the checks the reference's `with_ymd_and_hms(…).unwrap()` panics on: - * a year beyond ±262143, an impossible month or day, or a time past + * a year outside −262143…+262142, an impossible month or day, or a time past * 23:59:59 (no leap second here; `fromString` accepts `:60`). */ static fromYmdHms( @@ -313,17 +296,18 @@ export class CborDate implements CborTagged { ): CborDate { const seconds = civilSeconds(year, month, day, hour, minute, second); if (seconds === undefined) throw CborError.invalidDate("Invalid date components"); - const instance = new CborDate(); - instance._seconds = seconds; - return instance; + return new CborDate(seconds, 0); } /** - * Creates a new `CborDate` from seconds since (or before) the Unix epoch. + * Creates a new `CborDate` from seconds since the Unix epoch + * (1970-01-01T00:00:00Z); negative values are before the epoch. * - * This method creates a new `CborDate` representing the specified number of - * seconds since the Unix epoch (1970-01-01T00:00:00Z). Negative values - * represent times before the epoch. + * The value is split as the reference's `from_timestamp` splits it: whole + * seconds by truncation toward zero, then the fraction in nanoseconds + * (truncated, never negative), so `-1.5` is the instant `-1` and + * `1.0000000001` is `1`. `NaN` is the epoch, as the reference's saturating + * cast makes it. * * @param secondsSinceUnixEpoch - Seconds from the Unix epoch (positive or * negative), which can include a fractional part for sub-second @@ -331,6 +315,10 @@ export class CborDate implements CborTagged { * * @returns A new `CborDate` instance * + * @throws `InvalidDate` for ±Infinity, or when the whole seconds fall + * outside the reference's representable range (years −262143 to 262142), + * where the reference panics. + * * @example * ```typescript * // Create a date from a timestamp @@ -344,11 +332,8 @@ export class CborDate implements CborTagged { * ``` */ static fromEpochSeconds(secondsSinceUnixEpoch: number): CborDate { - const instance = new CborDate(); - // Normalize on construction so the stored value (and thus its encoding, - // equality, and ordering) is canonical. - instance._seconds = normalizeTimestampSeconds(secondsSinceUnixEpoch); - return instance; + const [seconds, nanoseconds] = timestampParts(secondsSinceUnixEpoch); + return new CborDate(seconds, nanoseconds); } /** @@ -367,9 +352,9 @@ export class CborDate implements CborTagged { * `+12023-02-08`), one- or two-digit month and day, with whitespace * allowed before each number (`2023-2-8`, ` 2023-02-08`). * - * The fraction is kept exactly: the stored timestamp is the reference's - * `timestamp()` — whole seconds plus nanoseconds over 10⁹ — so a decimal - * fraction encodes to the same bytes on both sides. + * The fraction is kept exactly as nanoseconds, so a decimal fraction + * encodes to the same bytes on both sides (`timestamp()`: whole seconds + * plus nanoseconds over 10⁹) and a leap second still displays as `:60`. * * @param value - A string containing a date or date-time in ISO-8601/RFC-3339 * format @@ -414,9 +399,7 @@ export class CborDate implements CborTagged { const offset = (sign === "+" ? 1 : -1) * (offsetHours * 3_600 + offsetMinutes * 60); const whole = civilSeconds(Number(y), Number(mo), Number(d), Number(h), Number(mi), second); if (whole === undefined) throw invalidDate(); - const instance = new CborDate(); - instance._seconds = whole - offset + nanoseconds / 1_000_000_000; - return instance; + return new CborDate(whole - offset, nanoseconds); } const ymd = YMD.exec(value); @@ -425,9 +408,7 @@ export class CborDate implements CborTagged { const year = sign === undefined ? Number(plainYear) : Number(`${sign}${signedYear}`); const whole = civilSeconds(year, Number(mo), Number(d), 0, 0, 0); if (whole === undefined) throw invalidDate(); - const instance = new CborDate(); - instance._seconds = whole; - return instance; + return new CborDate(whole, 0); } throw invalidDate(); @@ -469,12 +450,10 @@ export class CborDate implements CborTagged { } /** - * Returns the underlying JavaScript `Date` object. - * - * This method provides access to the wrapped JavaScript `Date` - * instance. + * Returns a new JavaScript `Date` for this instant (millisecond precision; + * sub-millisecond digits are lost). * - * @returns The wrapped `Date` instance + * @returns A new `Date` instance * * @example * ```typescript @@ -484,7 +463,7 @@ export class CborDate implements CborTagged { * ``` */ toDate(): Date { - return new Date(this._seconds * 1000); + return new Date(this.epochSeconds * 1000); } /** @@ -493,6 +472,9 @@ export class CborDate implements CborTagged { * represent times before the epoch; the fractional part is sub-second * precision. * + * This is the reference's `timestamp()`: whole seconds plus nanoseconds + * over 10⁹, computed in `f64`, and it is the value that goes on the wire. + * * @example * ```typescript * const date = CborDate.fromYmd(2023, 2, 8); @@ -500,7 +482,7 @@ export class CborDate implements CborTagged { * ``` */ get epochSeconds(): number { - return this._seconds; + return this._seconds + this._nanoseconds / 1_000_000_000; } /** @@ -554,16 +536,17 @@ export class CborDate implements CborTagged { } /** - * Implementation of the `CborTagged` interface for `CborDate`. + * The CBOR tags for `CborDate`: tag 1, the RFC 8949 epoch-based date/time. * - * This implementation specifies that `CborDate` values are tagged with CBOR tag 1, - * which is the standard CBOR tag for date/time values represented as seconds - * since the Unix epoch per RFC 8949. + * The tag carries whatever name the global tags store has for 1 at the + * time of the call (`tags_for_values` in the reference): `date` once + * `registerStandardTags()` has run, otherwise none. That name is what a + * `WrongTag` error prints as the expected tag. * - * @returns A vector containing tag 1 + * @returns An array containing tag 1 */ cborTags(): Tag[] { - return [Tag.from(TAG_EPOCH_DATE_TIME, "date")]; + return [getGlobalTagsStore().tagForValue(TAG_DATE) ?? Tag.from(TAG_DATE)]; } /** @@ -599,16 +582,18 @@ export class CborDate implements CborTagged { } /** - * Populates this `CborDate` in place from an untagged CBOR value, which - * must be a number (integer or floating-point) of seconds since the Unix - * epoch. The static `CborDate.fromUntaggedCbor` is the usual entry point; - * this instance form exists for reuse. + * Creates a `CborDate` from an untagged CBOR value, which must be a number + * (integer or floating-point) of seconds since the Unix epoch. The static + * `CborDate.fromUntaggedCbor` is the usual entry point; this instance form + * exists for the `CborTagged` protocol and returns a new instance. * * @param cbor - The untagged CBOR value * - * @returns this (populated in place) + * @returns The decoded date * - * @throws Error if the CBOR value is not a valid timestamp + * @throws `WrongType` for a non-numeric value, `OutOfRange` for an integer + * `f64` cannot hold exactly, `InvalidDate` beyond the representable + * range. A float `NaN` is the epoch, as in the reference. */ fromUntaggedCbor(cbor: Cbor): CborDate { let timestamp: number; @@ -644,20 +629,21 @@ export class CborDate implements CborTagged { throw CborError.wrongType(); } - // Normalize the decoded value so it re-encodes canonically (e.g. a tag-1 - // float of -1.5 decodes and re-encodes as integer -1). - this._seconds = normalizeTimestampSeconds(timestamp); - return this; + // Split as `from_timestamp` does, so e.g. a tag-1 float of -1.5 decodes + // and re-encodes as the integer -1. + return CborDate.fromEpochSeconds(timestamp); } /** - * Populates this `CborDate` in place from a tag-1 CBOR value. + * Creates a `CborDate` from a tag-1 CBOR value (the `CborTagged` + * protocol's instance form; returns a new instance). * * @param cbor - Tagged CBOR value * - * @returns this (populated in place) + * @returns The decoded date * - * @throws Error if the CBOR value has the wrong tag or cannot be decoded + * @throws {CborError} `WrongType` if the value is not tagged, `WrongTag` + * for a tag other than 1, or what `fromUntaggedCbor` throws for the content */ fromTaggedCbor(cbor: Cbor): CborDate { const expectedTags = this.cborTags(); @@ -673,8 +659,7 @@ export class CborDate implements CborTagged { * @returns New CborDate instance */ static fromTaggedCbor(cbor: Cbor): CborDate { - const instance = new CborDate(); - return instance.fromTaggedCbor(cbor); + return CborDate.EPOCH.fromTaggedCbor(cbor); } /** @@ -688,7 +673,10 @@ export class CborDate implements CborTagged { */ static get codec(): CborCodec { dateCodec ??= { - tags: [Tag.from(TAG_EPOCH_DATE_TIME, "date")], + // Resolved per access, not memoized: the name follows the global store. + get tags(): Tag[] { + return CborDate.EPOCH.cborTags(); + }, decode: (c: Cbor): CborDate => CborDate.fromTaggedCbor(c), encode: (value: CborDate): Cbor => value.taggedCbor(), }; @@ -696,16 +684,15 @@ export class CborDate implements CborTagged { } static fromUntaggedCbor(cbor: Cbor): CborDate { - const instance = new CborDate(); - return instance.fromUntaggedCbor(cbor); + return CborDate.EPOCH.fromUntaggedCbor(cbor); } + /** 1970-01-01T00:00:00Z: the receiver for the protocol's instance decoders. */ + private static readonly EPOCH = new CborDate(0, 0); + /** - * Implementation of the `toString` method for `CborDate`. - * - * This implementation provides a string representation of a `CborDate` in ISO-8601 - * format. For dates with time exactly at midnight (00:00:00), only the date - * part is shown. For other times, a full date-time string is shown. + * The date in ISO-8601 format: only the date part when the time is exactly + * midnight (00:00:00), otherwise a date-time to the second with `Z`. * * @returns String representation in ISO-8601 format * @@ -718,7 +705,7 @@ export class CborDate implements CborTagged { * * // A date with time will display as date and time * const date2 = CborDate.fromYmdHms(2023, 2, 8, 15, 30, 45); - * // Returns "2023-02-08T15:30:45.000Z" + * // Returns "2023-02-08T15:30:45Z" * console.log(date2.toString()); * ``` */ @@ -727,8 +714,9 @@ export class CborDate implements CborTagged { // (a fraction of a second does not count), otherwise RFC 3339 to the // second with `Z`. The year is four digits for 0–9999 and a sign plus at // least four digits beyond (`-0004`, `+12023`); JS `toISOString` would - // print six digits there. - const total = Math.floor(this._seconds); + // print six digits there. A leap second (nanoseconds >= 10⁹) prints as + // `:60`, as chrono formats it. + const total = this._seconds; const days = Math.floor(total / 86_400); const secondOfDay = total - days * 86_400; const [year, month, day] = civilFromDays(days); @@ -741,29 +729,36 @@ export class CborDate implements CborTagged { if (secondOfDay === 0) return date; const hour = Math.floor(secondOfDay / 3_600); const minute = Math.floor((secondOfDay % 3_600) / 60); - const second = secondOfDay % 60; + const second = (secondOfDay % 60) + (this._nanoseconds >= 1_000_000_000 ? 1 : 0); return `${date}T${pad(hour)}:${pad(minute)}:${pad(second)}Z`; } /** - * Compare two dates for equality. + * Compare two dates for equality: the same whole seconds and the same + * nanoseconds (chrono's `PartialEq`). A leap second `23:59:60` is a + * different instant from the following `00:00:00`, although both encode + * to the same wire value. * * @param other - Other CborDate to compare * @returns true if dates represent the same moment in time */ equals(other: CborDate): boolean { - return this._seconds === other._seconds; + return this._seconds === other._seconds && this._nanoseconds === other._nanoseconds; } /** - * Compare two dates. + * Compare two dates: by whole seconds, then by nanoseconds (chrono's + * `Ord`, so a leap second sorts after `:59.999999999` and before the next + * `:00`). * * @param other - Other CborDate to compare * @returns -1 if this < other, 0 if equal, 1 if this > other */ compare(other: CborDate): number { - if (this._seconds < other._seconds) return -1; - if (this._seconds > other._seconds) return 1; + if (this._seconds !== other._seconds) return this._seconds < other._seconds ? -1 : 1; + if (this._nanoseconds !== other._nanoseconds) { + return this._nanoseconds < other._nanoseconds ? -1 : 1; + } return 0; } @@ -776,7 +771,8 @@ export class CborDate implements CborTagged { return this.toString(); } - private constructor() { - this._seconds = Date.now() / 1000; + private constructor(seconds: number, nanoseconds: number) { + this._seconds = seconds; + this._nanoseconds = nanoseconds; } } diff --git a/src/decode.ts b/src/decode.ts index cdc5771..2ec9fec 100644 --- a/src/decode.ts +++ b/src/decode.ts @@ -1,10 +1,28 @@ import { type Cbor, encodeCbor, attachMethods } from "./cbor"; import { type CborNumber, MajorType } from "./cbor-types"; import { areBytesEqual } from "./stdlib"; -import { binary16ToNumber, binary32ToNumber, binary64ToNumber } from "./float"; +import { + binary16ToNumber, + binary32ToNumber, + binary64ToNumber, + cborNodeFromF16, + cborNodeFromF32, + cborNodeFromF64, + validateCanonicalF16, + validateCanonicalF32, + validateCanonicalF64, +} from "./float"; import { CborMap } from "./map"; import { CborError, Ok, Err, type Result } from "./error"; import { narrowInteger } from "./numeric"; +import { utf8ErrorDescription } from "./utf8"; + +// One reused strict UTF-8 decoder. `fatal` rejects malformed sequences instead +// of substituting U+FFFD; `ignoreBOM` keeps a leading U+FEFF as a character - +// the WHATWG default strips it, whereas Rust's `String::from_utf8` (the +// reference decoder) preserves every code point, so a text string starting +// with a BOM must survive a decode -> re-encode round trip byte-for-byte. +const utf8Decoder = new TextDecoder("utf-8", { fatal: true, ignoreBOM: true }); /** * A forward-only cursor over the input bytes. @@ -65,8 +83,7 @@ class ByteReader { * @remarks Decoded byte strings are zero-copy views aliasing the input * buffer - mutating the input after decoding (or mutating the returned * bytes) changes the other side. Call `.slice()` first if you need an - * independent copy. This is deliberate: the zero-copy decode performance - * profile is part of the library's contract. + * independent copy. */ export function decodeCbor(data: Uint8Array): Cbor { const reader = new ByteReader(data); @@ -81,7 +98,8 @@ export function decodeCbor(data: Uint8Array): Cbor { /** * Decode without throwing: returns a {@link Result} carrying the decoded value, * or the {@link CborError} that {@link decodeCbor} would have thrown. Non-CBOR - * errors still propagate. + * errors still propagate - including the host's `RangeError` when a deeply + * nested input exhausts the call stack (the reference aborts there too). * * The `try` prefix means "returns `Result`, never throws" - everywhere in * this library. @@ -230,14 +248,16 @@ function readCbor(reader: ByteReader): Cbor { } const textBytes = reader.bytesAt(reader.pos, value); reader.advance(value); - // dCBOR text strings must be valid UTF-8 (RFC 8949). Use a fatal decoder - // so invalid bytes throw rather than getting replaced with U+FFFD and - // silently accepted. + // dCBOR text strings must be valid UTF-8 (RFC 8949). The decoder is + // fatal so invalid bytes throw rather than getting replaced with U+FFFD + // and silently accepted. The host's error text is discarded: the + // message is the reference's `Utf8Error` description, computed from + // the bytes. let text: string; try { - text = new TextDecoder("utf-8", { fatal: true }).decode(textBytes); - } catch (e) { - throw CborError.invalidUtf8(e instanceof Error ? e.message : String(e)); + text = utf8Decoder.decode(textBytes); + } catch { + throw CborError.invalidUtf8(utf8ErrorDescription(textBytes)); } // dCBOR requires all text strings to be in Unicode Normalization Form C // (NFC); reject any that are not already normalized. @@ -272,38 +292,28 @@ function readCbor(reader: ByteReader): Cbor { } as const); } case MajorType.Simple: + // Float heads: the reference's `validate_canonical_f*` predicates decide + // acceptance and its `From` impls build the node - not a re-encode + // comparison. The two differ for whole-valued heads at or beyond the + // saturating-cast bounds (2^31 for f32, 2^63 for f64): those are + // accepted and reduce to integer nodes, so e.g. `fa4f000001` decodes to + // the unsigned 2147483904 and re-encodes as `1a80000100`. switch (varIntLen) { case 3: { + // `value` is the raw 16-bit pattern; the canonical NaN is 0x7e00. const f = binary16ToNumber(reader.bytesAt(headStart + 1, 2)); - // dCBOR canonical-encoding check via re-encode-and-compare. JS's - // `Number` type does not preserve NaN payload bits - every NaN - // collapses to the same value - so a bit-level check is not possible. - // Re-encoding round-trips through the canonicalising encoder, catching - // every non-canonical NaN, ±Infinity, and integer-reducible float. - checkCanonicalEncoding(f, reader.bytesAt(headStart, varIntLen)); - return attachMethods({ - isCbor: true, - type: MajorType.Simple, - value: { type: "Float", value: f }, - } as const); + validateCanonicalF16(Number(value), f); + return attachMethods(cborNodeFromF16(f)); } case 5: { const f = binary32ToNumber(reader.bytesAt(headStart + 1, 4)); - checkCanonicalEncoding(f, reader.bytesAt(headStart, varIntLen)); - return attachMethods({ - isCbor: true, - type: MajorType.Simple, - value: { type: "Float", value: f }, - } as const); + validateCanonicalF32(f); + return attachMethods(cborNodeFromF32(f)); } case 9: { const f = binary64ToNumber(reader.bytesAt(headStart + 1, 8)); - checkCanonicalEncoding(f, reader.bytesAt(headStart, varIntLen)); - return attachMethods({ - isCbor: true, - type: MajorType.Simple, - value: { type: "Float", value: f }, - } as const); + validateCanonicalF64(f); + return attachMethods(cborNodeFromF64(f)); } default: switch (value) { @@ -333,8 +343,7 @@ function readCbor(reader: ByteReader): Cbor { } } -function checkCanonicalEncoding(cbor: Cbor | CborNumber, buf: Uint8Array): void { - // encodeCbor accepts both decoded CBOR objects and native values (floats). +function checkCanonicalEncoding(cbor: Cbor, buf: Uint8Array): void { const buf2 = encodeCbor(cbor); if (!areBytesEqual(buf, buf2)) { throw CborError.nonCanonicalNumeric(); diff --git a/src/diag.ts b/src/diag.ts index 0eeceff..5fde274 100644 --- a/src/diag.ts +++ b/src/diag.ts @@ -1,7 +1,7 @@ /** - * Enhanced diagnostic formatting for CBOR values. + * Diagnostic notation formatting for CBOR values. * - * Provides multiple formatting options including + * Formatting options: * - Annotated diagnostics with tag names * - Summarized values using custom summarizers * - Flat (single-line) vs. pretty (multi-line) formatting @@ -26,7 +26,8 @@ import { flanked } from "./string-util"; export interface DiagFormatOpts { /** * Add tag names as annotations. - * When true, tagged values are displayed as "tagName(content)" instead of "tagValue(content)". + * When true, a tagged value whose tag has a name in the store is followed by + * a `/ name /` comment, e.g. `1(1675854714) / date /`. * * @default false */ @@ -34,7 +35,7 @@ export interface DiagFormatOpts { /** * Use custom summarizers for tagged values. - * When true, calls registered summarizers for tagged values. + * When true, calls registered summarizers for tagged values. Implies `flat`. * * @default false */ @@ -102,7 +103,6 @@ const resolveOpts = (opts?: DiagFormatOpts): DiagState => { */ export function diagnostic(input: Cbor | WalkElement, opts?: DiagFormatOpts): string { const state = resolveOpts(opts); - // WalkElement support is load-bearing for walk visitors. if ( typeof input === "object" && "type" in input && @@ -186,8 +186,8 @@ const greatestStringsLen = (i: DiagItem): number => /** * Alternates between `pairSeparator` (after even-indexed items - keys) and - * `itemSeparator` (after odd-indexed items - values). Falls back to - * `itemSeparator` for non-pair groups. + * `itemSeparator` (after odd-indexed items - values). Uses `itemSeparator` + * throughout when `pairSeparator` is omitted. */ function joined(elements: string[], itemSeparator: string, pairSeparator?: string): string { const sep = pairSeparator ?? itemSeparator; @@ -328,8 +328,8 @@ function item_tagged(tag: number | bigint, content: Cbor, opts: DiagFormatOpts): if (result.ok) { return item(result.value); } - // Use the shared error formatter so every variant gets its full message, - // including name-aware tag rendering for WrongTag. + // The error's message is its full text, including the tag names of a + // WrongTag. return item(``); } } @@ -347,7 +347,6 @@ function item_tagged(tag: number | bigint, content: Cbor, opts: DiagFormatOpts): return group(`${String(tag)}(`, ")", [diagItem(content, opts)], false, comment); } -// Primitive formatters reused by both single- and multi-line paths. function formatUnsigned(value: number | bigint): string { return String(value); } @@ -382,8 +381,8 @@ function formatSimple(value: Simple): string { } /** - * Format a CBOR float for diagnostic output. Shared with the hex-dump - * annotation path; see {@link floatDisplayString}. + * Format a CBOR float for diagnostic output, with the same rendering + * `hexAnnotated` uses; see {@link floatDisplayString}. */ function formatFloat(value: number): string { return floatDisplayString(value); diff --git a/src/diagnostic.ts b/src/diagnostic.ts index a91fda6..9207a9d 100644 --- a/src/diagnostic.ts +++ b/src/diagnostic.ts @@ -3,7 +3,7 @@ * * Human-readable rendering of CBOR values: diagnostic notation and annotated * hex dumps. Kept out of the root entry so decode-only bundles never carry - * the formatter, tag store retainers, or the walker. + * the formatters. * * ```typescript * import { diagnostic, hexAnnotated } from "@blockchaincommons/dcbor/diagnostic"; diff --git a/src/dump.ts b/src/dump.ts index f915a07..8b824bb 100644 --- a/src/dump.ts +++ b/src/dump.ts @@ -19,6 +19,18 @@ import { Tag } from "./tag"; import { CborError } from "./error"; import { bytesToHex } from "./hex"; +// Strict UTF-8 decoder for the byte-string annotation. `ignoreBOM` keeps a +// leading U+FEFF so the note shows every code point the bytes carry, as the +// reference's `String::from_utf8` does. +const utf8Decoder = new TextDecoder("utf-8", { fatal: true, ignoreBOM: true }); + +// One reused UTF-8 encoder for the text-string lines (TextEncoder is +// stateless). It encodes the node's string as stored, without the NFC pass the +// encoder applies, because the reference's `dump_items` uses `s.as_bytes()` +// rather than `to_cbor_data()` - a node built from a non-NFC string dumps its +// raw bytes on both sides. +const utf8Encoder = new TextEncoder(); + /** * Options for annotated hex formatting. */ @@ -51,7 +63,8 @@ export const hexAnnotated = (cbor: Cbor, opts?: HexFormatOpts): string => { return Math.max(largest, item.formatFirstColumn().length); }, 0); - // Round up to nearest multiple of 4 + // One less than the next multiple of 4 above the widest first column, as + // the reference's `hex_opt` computes it. const roundedNoteColumn = ((noteColumn + 4) & ~3) - 1; const lines = items.map((item) => item.format(roundedNoteColumn)); @@ -118,9 +131,9 @@ function dumpItems(cbor: Cbor, level: number, tagsStore: TagsStore): DumpItem[] let note: string | undefined = undefined; // Try to decode as UTF-8 string for annotation try { - const text = new TextDecoder("utf-8", { fatal: true }).decode(cbor.value); + const text = utf8Decoder.decode(cbor.value); const sanitizedText = sanitized(text); - if (sanitizedText !== undefined && sanitizedText !== "") { + if (sanitizedText !== undefined) { note = flanked(sanitizedText, '"', '"'); } } catch { @@ -133,7 +146,7 @@ function dumpItems(cbor: Cbor, level: number, tagsStore: TagsStore): DumpItem[] } case MajorType.Text: { - const utf8Data = new TextEncoder().encode(cbor.value); + const utf8Data = utf8Encoder.encode(cbor.value); const header = encodeVarInt(utf8Data.length, MajorType.Text); const firstByte = header[0]; if (firstByte === undefined) { diff --git a/src/error.ts b/src/error.ts index a389174..1ab3e8b 100644 --- a/src/error.ts +++ b/src/error.ts @@ -103,8 +103,8 @@ export interface CborErrorDetailsByCode { * } * ``` * - * The intersection with the legacy {@link CborErrorDetails} bag keeps - * un-narrowed `details` access compiling exactly as before. + * The intersection with the deprecated {@link CborErrorDetails} bag keeps + * every detail field readable (as optional) on an un-narrowed error. */ export type CborErrorTyped = C extends CborErrorCode ? CborError & { diff --git a/src/exact.ts b/src/exact.ts index c35125b..8339fc4 100644 --- a/src/exact.ts +++ b/src/exact.ts @@ -16,7 +16,6 @@ import { binary16ToNumber, binary32ToNumber, numberToBinary16, numberToBinary32 // TypeScript doesn't have native integer types with overflow, so we use number for most operations // and bigint for values that exceed Number.MAX_SAFE_INTEGER -// Helper to check if a number has a fractional part const hasFract = (n: number): boolean => { return n % 1 !== 0; }; @@ -54,6 +53,9 @@ export class ExactI16 { static readonly MIN = -32768; static readonly MAX = 32767; + // The reference's `exact_from_f16` for i16 excludes -32768 (`source <= + // -32768.0`), although binary16 represents it; the f32 and f64 forms + // accept it. Kept identical (RUST_DIVERGENCES.md §2). static exactFromF16(source: number): number | undefined { return intFromFloatNum(source, -32768.0, 32768.0); } @@ -383,8 +385,8 @@ export class ExactU128 { } /** - * Exact conversions for f16 (half precision float). - * In TypeScript, we work with f16 as raw bytes (Uint8Array of length 2). + * Exact conversions for f16 (half precision float). An f16 is a JS `number` + * that is exactly representable in binary16. */ export class ExactF16 { static exactFromF16(source: number): number | undefined { diff --git a/src/extract.ts b/src/extract.ts index fc55e4f..ec1549c 100644 --- a/src/extract.ts +++ b/src/extract.ts @@ -2,9 +2,8 @@ * Native-value extraction from CBOR. * * Provides {@link extractCbor}, which converts a CBOR value into its native - * JavaScript equivalent. Kept as a small leaf module so that consumers needing - * only extraction do not pull in the whole convenience surface, keeping the - * module graph acyclic. + * JavaScript equivalent. A separate module so `CborMap` and `CborSet` can + * use it without importing the convenience surface. * * @module extract */ @@ -28,11 +27,9 @@ export type CborNative = number | bigint | string | boolean | null | Uint8Array | CborNative[] | CborMap | Cbor; /** - * Extract native JavaScript value from CBOR. - * Converts CBOR types to their JavaScript equivalents. - * - * Returns the closed union {@link CborNative}. Note the two asymmetries - * documented there: maps come back as `CborMap` and tagged values as `Cbor`. + * Extract the native JavaScript value from a CBOR value, decoding it first + * when given bytes. Maps come back as `CborMap` and tagged values as `Cbor` + * (see {@link CborNative}). */ export const extractCbor = (cbor: Cbor | Uint8Array): CborNative => { let c: Cbor; diff --git a/src/float.ts b/src/float.ts index 4230500..c9ef2e3 100644 --- a/src/float.ts +++ b/src/float.ts @@ -1,28 +1,26 @@ /** * Float encoding and conversion utilities for dCBOR. * - * # Floating Point Number Support in dCBOR + * The dCBOR canonical encoding rules for floating point values: * - * dCBOR provides canonical encoding for floating point values. - * - * Per the dCBOR specification, the canonical encoding rules ensure - * deterministic representation: - * - * - Numeric reduction: Floating point values with zero fractional part in - * range [-2^63, 2^64-1] are automatically encoded as integers (e.g., 42.0 - * becomes 42) - * - Values are encoded in the smallest possible representation that preserves - * their value - * - All NaN values are canonicalized to a single representation: 0xf97e00 - * - Positive/negative infinity are canonicalized to half-precision - * representations + * - Numeric reduction: a float with zero fractional part in + * [-2^64, 2^64-1] is encoded as an integer (42.0 becomes 42) + * - Other values use the smallest width (f16, f32, f64) that preserves them + * - Every NaN is encoded as the single representation 0xf97e00 + * - Positive and negative infinity are encoded as half-precision floats * * @module float */ import { encodeVarInt } from "./varint"; -import { MajorType } from "./cbor-types"; +import { + type CborNegativeType, + type CborSimpleType, + type CborUnsignedType, + MajorType, +} from "./cbor-types"; import { ExactU64, ExactU32, ExactU16, ExactI128 } from "./exact"; +import { CborError } from "./error"; /** * Canonical NaN representation in CBOR: 0xf97e00 @@ -65,11 +63,10 @@ const f32ScratchView = new DataView(new ArrayBuffer(4)); * Compute the 16-bit pattern of the IEEE-754 half-precision value nearest `n`, * rounding ties to even. * - * All call sites pass values already exactly representable in binary16 (the - * reduction gates in {@link f16CborData} ensure this), so no rounding occurs on - * a value that is actually stored; the rounding path exists only so the - * reduction round-trip probe (`binary16ToNumber(numberToBinary16(n)) === n`) - * answers correctly for non-representable inputs. + * A value is only stored as a half after the round-trip probe + * (`binary16ToNumber(numberToBinary16(n)) === n`) succeeds, so stored values + * never round; the rounding makes that probe, and the reference's + * `f16::from_f32` in `validateCanonicalF32`, answer correctly for any input. */ const float16Bits = (n: number): number => { f32ScratchView.setFloat32(0, n, false); @@ -259,38 +256,254 @@ export const f16CborData = (value: number): Uint8Array => { return new Uint8Array([0xf9, ...bytes]); }; +// ============================================================================ +// Decoder-side canonicality predicates and node construction +// +// Ports of `validate_canonical_f16/f32/f64` and `From for CBOR` +// in the reference's float.rs. The predicates use Rust's saturating `as` +// casts on purpose: a whole-valued f32 head at or beyond 2^31 (or an f64 head +// at or beyond 2^63) does NOT compare equal to its saturated integer image, +// so the reference accepts it, and its `From` impl then reduces it to an +// integer node where one fits. The decoder must reproduce that accept set +// and those nodes byte-for-byte. +// ============================================================================ + +const TWO_POW_63 = 2 ** 63; + +/** + * Rust `n as i64 as f64`: NaN → 0; saturates at the i64 bounds. `i64::MAX` + * (2^63 - 1) is not a double, so the saturated image converts back to 2^63, + * and every double at or above 2^63 saturates to it. + */ +const saturatingI64AsF64 = (n: number): number => { + if (Number.isNaN(n)) return 0; + if (n >= TWO_POW_63) return TWO_POW_63; // i64::MAX as f64 + if (n <= -TWO_POW_63) return -TWO_POW_63; // i64::MIN + return Math.trunc(n); +}; + +/** Rust `n as i32 as f32` for an f32 value: NaN → 0; saturates at the i32 bounds. */ +const saturatingI32AsF32 = (n: number): number => { + if (Number.isNaN(n)) return 0; + if (n >= 2147483647) return Math.fround(2147483647); // i32::MAX as f32 = 2^31 + if (n <= -2147483648) return -2147483648; // i32::MIN + return Math.fround(Math.trunc(n)); +}; + +/** + * `validate_canonical_f16`: a half head is non-canonical when it is + * whole-valued (must be an integer) or a NaN other than `0x7e00`. + * @internal + */ +export const validateCanonicalF16 = (bits: number, n: number): void => { + if (n === saturatingI64AsF64(n) || (Number.isNaN(n) && bits !== 0x7e00)) { + throw CborError.nonCanonicalNumeric(); + } +}; + +/** + * `validate_canonical_f32`: a single head is non-canonical when it fits a half + * (including ±0 and ±Infinity), equals its saturating `i32` image, or is NaN. + * @internal + */ +export const validateCanonicalF32 = (n: number): void => { + if ( + n === binary16ToNumber(numberToBinary16(n)) || + n === saturatingI32AsF32(n) || + Number.isNaN(n) + ) { + throw CborError.nonCanonicalNumeric(); + } +}; + +/** + * `validate_canonical_f64`: a double head is non-canonical when it fits a + * single, equals its saturating `i64` image, or is NaN. + * @internal + */ +export const validateCanonicalF64 = (n: number): void => { + if (n === Math.fround(n) || n === saturatingI64AsF64(n) || Number.isNaN(n)) { + throw CborError.nonCanonicalNumeric(); + } +}; + +/** A bare (methodless) node a float head decodes to. */ +export type FloatHeadNode = CborUnsignedType | CborNegativeType | CborSimpleType; + +const unsignedNode = (value: number | bigint): CborUnsignedType => ({ + isCbor: true, + type: MajorType.Unsigned, + value, +}); +const negativeNode = (magnitude: number | bigint): CborNegativeType => ({ + isCbor: true, + type: MajorType.Negative, + value: magnitude, +}); +const floatNode = (value: number): CborSimpleType => ({ + isCbor: true, + type: MajorType.Simple, + value: { type: "Float", value }, +}); + +/** `From for CBOR`. @internal */ +export const cborNodeFromF16 = (n: number): FloatHeadNode => { + if (n < 0) { + const i = ExactU64.exactFromF64(-1 - n); + if (i !== undefined) return negativeNode(i); + } + const u = ExactU16.exactFromF64(n); + if (u !== undefined) return unsignedNode(u); + return floatNode(n); +}; + +/** + * `From for CBOR`. The negative magnitude is computed in f32 arithmetic + * (`-1f32 - n`): `Math.fround(-1 - n)` is exactly that, since a double holds + * the difference of two singles with at most one rounding. + * @internal + */ +export const cborNodeFromF32 = (n: number): FloatHeadNode => { + if (n < 0) { + const i = ExactU64.exactFromF32(Math.fround(-1 - n)); + if (i !== undefined) return negativeNode(i); + } + const u = ExactU32.exactFromF32(n); + if (u !== undefined) return unsignedNode(u); + return floatNode(n); +}; + +/** `From for CBOR`. @internal */ +export const cborNodeFromF64 = (n: number): FloatHeadNode => { + if (n < 0) { + const i128 = ExactI128.exactFromF64(n); + if (i128 !== undefined) { + const i = ExactU64.exactFromI128(-1n - i128); + if (i !== undefined) return negativeNode(i); + } + } + const u = ExactU64.exactFromF64(n); + if (u !== undefined) return unsignedNode(u); + return floatNode(n); +}; + /** - * Render a float to its diagnostic string. + * Shortest round-trip decimal digits of a finite positive double, as the pair + * (significant digits without trailing zeros, scientific exponent), where the + * value is `d1.d2…dk × 10^exp10`. * - * Finite non-zero values with magnitude in [1e-4, 1e16) print in decimal with - * at least one fractional digit (whole values get a trailing `.0`); everything - * else prints in exponential form. Zero prints as `0.0`/`-0.0`. + * `String(x)` already yields the shortest digit string; this only re-shapes + * it (ECMAScript picks between "123.45", "1.5e-7", "1e+21" and "0.000001" by + * magnitude) so the caller can apply Rust's notation rules. + */ +const shortestDigits = (abs: number): { digits: string; exp10: number } => { + const text = String(abs); + const eIndex = text.indexOf("e"); + const mantissa = eIndex === -1 ? text : text.slice(0, eIndex); + const exponent = eIndex === -1 ? 0 : Number(text.slice(eIndex + 1)); + const dot = mantissa.indexOf("."); + let digits = dot === -1 ? mantissa : mantissa.slice(0, dot) + mantissa.slice(dot + 1); + let pointPos = dot === -1 ? mantissa.length : dot; + while (digits.length > 1 && digits.startsWith("0")) { + digits = digits.slice(1); + pointPos--; + } + digits = digits.replace(/0+$/, ""); + if (digits === "") digits = "0"; + return { digits, exp10: pointPos - 1 + exponent }; +}; + +/** The exact value of a finite positive double as `mantissa × 2^exp2`. */ +const exactBinary = (abs: number): { mantissa: bigint; exp2: number } => { + const view = new DataView(new ArrayBuffer(8)); + view.setFloat64(0, abs, false); + const hi = view.getUint32(0, false); + const lo = view.getUint32(4, false); + const biasedExp = (hi >>> 20) & 0x7ff; + const fraction = (BigInt(hi & 0xfffff) << 32n) | BigInt(lo); + return biasedExp === 0 + ? { mantissa: fraction, exp2: -1074 } + : { mantissa: fraction | (1n << 52n), exp2: biasedExp - 1075 }; +}; + +/** + * The exact decimal expansion of a finite positive double, as its significant + * digits (no trailing zeros). Every double is a dyadic rational, so + * `m × 2^q = m × 5^-q / 10^-q` for negative `q` gives the digits exactly. + */ +const exactDecimalDigits = (abs: number): string => { + const { mantissa, exp2 } = exactBinary(abs); + const scaled = exp2 >= 0 ? mantissa << BigInt(exp2) : mantissa * 5n ** BigInt(-exp2); + return scaled.toString().replace(/0+$/, ""); +}; + +/** + * Shortest round-trip digits the way Rust's `{:?}` produces them. * - * JS already produces the same shortest round-tripping digits; we only fix up - * the notation threshold, the `e+` → `e` exponent, and the `.0` suffix. + * JS and Rust agree on the shortest digit string except when the exact value + * sits precisely halfway between the two shortest candidates: ECMAScript + * (`Number::toString`) picks the even candidate, Rust's `flt2dec` rounds the + * magnitude up. `10 × 2^-24` is exactly `5.9604644775390625e-7`, which JS + * prints as `…062e-7` and Rust as `…063e-7`. Detect the tie exactly and take + * the upper candidate when it also round-trips. + */ +const rustShortestDigits = (abs: number): { digits: string; exp10: number } => { + const shortest = shortestDigits(abs); + const k = shortest.digits.length; + // Cheap filter: a tie means the (k+1)-digit rounding ends in exactly 5. + const probe = abs.toPrecision(k + 1); + const probeIndex = probe.indexOf("e"); + if (!(probeIndex === -1 ? probe : probe.slice(0, probeIndex)).endsWith("5")) return shortest; + const exact = exactDecimalDigits(abs); + if (exact.length !== k + 1 || !exact.endsWith("5")) return shortest; + // Exact tie: the upper candidate is the truncated expansion plus one unit. + let upper = (BigInt(exact.slice(0, k)) + 1n).toString(); + let exp10 = shortest.exp10; + if (upper.length > k) { + // A carry (…999 + 1) shifts the decimal point. + exp10 += 1; + } + upper = upper.replace(/0+$/, ""); + if (upper === "") upper = "0"; + const candidate = Number(`${upper[0]}.${upper.slice(1)}e${exp10}`); + return candidate === abs ? { digits: upper, exp10 } : shortest; +}; + +/** + * Render a float to its diagnostic string, the reference's `Display for + * Simple` (`simple.rs`) - the rendering `diagnostic()` and `hexAnnotated()` + * use. + * + * Non-finite values print as `NaN`, `Infinity` and `-Infinity`, exactly as the + * reference's `Display` does - not the `inf`/`-inf` of Rust's `{:?}`, which is + * `Simple::name()`'s rendering and is ported as `simpleName` (`simple.ts`). * - * @param value - The float value - * @returns The diagnostic string + * Finite values match Rust's `{:?}` for `f64`: non-zero values with magnitude + * in [1e-4, 1e16) print in decimal with at least one fractional digit (whole + * values get a trailing `.0`); everything else prints in exponential form + * (`1.5e20`, `5e-324` - no `+`, no padding). Zero prints as `0.0`/`-0.0`. + * Digits are the shortest round-trip sequence, with exact decimal ties rounded + * up like Rust (see {@link rustShortestDigits}). */ export const floatDisplayString = (value: number): string => { if (Number.isNaN(value)) return "NaN"; if (!Number.isFinite(value)) return value > 0 ? "Infinity" : "-Infinity"; if (value === 0) return Object.is(value, -0) ? "-0.0" : "0.0"; - const abs = Math.abs(value); - if (abs >= 1e-4 && abs < 1e16) { - // In this range String() never switches to exponential. Ensure at least - // one fractional digit. - let str = String(value); - if (!str.includes(".")) { - str = `${str}.0`; + const sign = value < 0 ? "-" : ""; + const { digits, exp10 } = rustShortestDigits(Math.abs(value)); + + if (exp10 >= -4 && exp10 < 16) { + if (exp10 < 0) { + return `${sign}0.${"0".repeat(-exp10 - 1)}${digits}`; } - return str; + const intLen = exp10 + 1; + const intPart = digits.length >= intLen ? digits.slice(0, intLen) : digits.padEnd(intLen, "0"); + const fracPart = digits.length > intLen ? digits.slice(intLen) : "0"; + return `${sign}${intPart}.${fracPart}`; } - - // Drop the `+` in the exponent (`1.5e+20` → `1.5e20`); negative exponents - // keep their sign. - return value.toExponential().replace("e+", "e"); + const mantissa = digits.length > 1 ? `${digits[0]}.${digits.slice(1)}` : digits; + return `${sign}${mantissa}e${exp10}`; }; /** diff --git a/src/hex.ts b/src/hex.ts index 9e5d661..76be302 100644 --- a/src/hex.ts +++ b/src/hex.ts @@ -50,8 +50,7 @@ export const hexToBytes = (hexString: string): Uint8Array => { throw CborError.custom("invalid hex string"); } if (nativeFromHex !== undefined) { - // Native fromHex only accepts lowercase+uppercase hex, which the - // validation above guarantees. + // Native fromHex rejects whitespace; it was stripped above. return nativeFromHex(hex); } const bytes = new Uint8Array(hex.length / 2); diff --git a/src/index.ts b/src/index.ts index 7d3b3f7..4f3fc32 100644 --- a/src/index.ts +++ b/src/index.ts @@ -35,7 +35,7 @@ export { type CborMapType, type CborTaggedType, type CborSimpleType, - // The ONE structural conversion protocol accepted by `cbor()`. + // The structural conversion protocol accepted by `cbor()`. type ToCbor, } from "./cbor"; @@ -67,31 +67,12 @@ export { } from "./tags-store"; // The standard CBOR tags this library defines. export { - TAG_DATE_TIME_STRING, - TAG_EPOCH_DATE_TIME, - TAG_EPOCH_DATE, + TAG_DATE, + TAG_NAME_DATE, TAG_POSITIVE_BIGNUM, TAG_NEGATIVE_BIGNUM, TAG_NAME_POSITIVE_BIGNUM, TAG_NAME_NEGATIVE_BIGNUM, - TAG_DECIMAL_FRACTION, - TAG_BIGFLOAT, - TAG_BASE64URL, - TAG_BASE64, - TAG_BASE16, - TAG_ENCODED_CBOR, - TAG_URI, - TAG_BASE64URL_TEXT, - TAG_BASE64_TEXT, - TAG_REGEXP, - TAG_MIME_MESSAGE, - TAG_UUID, - TAG_STRING_REF_NAMESPACE, - TAG_BINARY_UUID, - TAG_SET, - TAG_SELF_DESCRIBE_CBOR, - TAG_DATE, - TAG_NAME_DATE, registerStandardTags, type RegisterStandardTagsOptions, tagsForValues, @@ -174,6 +155,7 @@ export { // Convenience utilities - expectations (`expect*` = `T` or throw CborError) export { expectUnsigned, + type ExpectUnsignedOptions, expectNegative, expectInteger, expectBytes, diff --git a/src/map.ts b/src/map.ts index 7bec5f0..28279dd 100644 --- a/src/map.ts +++ b/src/map.ts @@ -1,26 +1,15 @@ /** - * Map Support in dCBOR + * A deterministic CBOR map: maps with the same content encode identically, + * regardless of insertion order. * - * A deterministic CBOR map implementation that ensures maps with the same - * content always produce identical binary encodings, regardless of insertion - * order. - * - * ## Deterministic Map Representation - * - * The `CborMap` type follows strict deterministic encoding rules as specified by - * dCBOR: - * - * - Map keys are always sorted in lexicographic order of their encoded CBOR bytes - * - Duplicate keys are not allowed (enforced by the implementation) + * - Entries are kept in lexicographic order of their encoded key bytes + * - Setting a key whose encoding is already present replaces that entry * - Keys and values can be any type that can be converted to CBOR - * - Numeric reduction is applied (e.g., 3.0 is stored as integer 3) - * - * ## Vocabulary * * `CborMap` mirrors the JS `Map` protocol: `set`, `get`, `getOrThrow`, `has`, * `delete`, `clear`, `size`, `keys()`, `values()`, `entries()`, `forEach`, - * iteration. `get` returns the STORED `Cbor` node (symmetric with - * `entries()`); extract natives explicitly with `extractCbor(map.get(k))`. + * iteration. `get` returns the stored `Cbor` node, like `entries()`; extract + * natives explicitly with `extractCbor(map.getOrThrow(k))`. * * @module map */ @@ -97,10 +86,9 @@ export class CborMap { } /** - * Get the STORED `Cbor` node for a key, or `undefined` if absent. + * Get the stored `Cbor` node for a key, or `undefined` if absent. * - * This is symmetric with `entries()` - no hidden native extraction, no - * unwitnessed generics. To read a native value, compose explicitly: + * To read a native value, compose explicitly: * * ```typescript * asNumber(map.get("age")); // number | undefined, checked @@ -160,6 +148,30 @@ export class CborMap { })); } + /** + * The stored entry at position `i` in canonical ascending encoded-key + * order; the caller keeps `i` within `[0, size)`. + * + * @internal Positional access for the encoder and for structural equality, + * which walk a map (or two maps in lockstep) without materializing + * `entriesArray`; not part of the supported surface. + */ + entryAt(i: number): MapEntry { + return this._dict.valueAt(i); + } + + /** + * The encoded CBOR bytes of the key at position `i` - the bytes the entry + * is sorted by, computed once when it was inserted. + * + * @internal The encoder writes these directly, as the reference's + * `Map::cbor_data` writes its stored `MapKey`, instead of re-encoding the + * key node; not part of the supported surface. + */ + encodedKeyAt(i: number): Uint8Array { + return this._dict.keyAt(i); + } + /** Iterate keys in canonical (sorted encoded-key) order. */ *keys(): Generator { for (const entry of this.entriesArray) { @@ -198,11 +210,12 @@ export class CborMap { } /** - * Inserts the next key-value pair into the map during decoding. - * This is used for efficient map building during CBOR decoding. - * Throws if the key is not in ascending order or is a duplicate. + * Append a key-value pair whose encoded key must sort strictly after every + * existing key. * * @internal The decoder's append path; not part of the supported surface. + * @throws {CborError} `DuplicateMapKey` for a repeated key, + * `MisorderedMapKey` for a key out of ascending order. */ setNext(key: CborInput, value: CborInput): void { const keyCbor = cbor(key); diff --git a/src/numeric.ts b/src/numeric.ts index d44fe1e..3f0b752 100644 --- a/src/numeric.ts +++ b/src/numeric.ts @@ -4,7 +4,7 @@ * ## The `number` / `bigint` contract * * dCBOR integers span `[-(2^64), 2^64)`, which exceeds JavaScript's safe - * integer range (`±(2^53 − 1)`). The single, repo-wide rule is: + * integer range (`±(2^53 − 1)`). The rule is: * * - An integer that fits in the IEEE-754 **safe** range is represented as a * `number`; anything larger (in magnitude) is a `bigint`. @@ -13,8 +13,7 @@ * ones remain lossless `bigint`s. * - Encoding accepts either at the public edge and normalises once. * - * Every module funnels its boundary logic through this file - nothing else - * should hard-code `Number.MAX_SAFE_INTEGER`, `2^64`, etc. + * The integer range constants and the saturating float casts live here. * * @module numeric */ diff --git a/src/set.ts b/src/set.ts index 8ced977..c5f8b48 100644 --- a/src/set.ts +++ b/src/set.ts @@ -52,9 +52,8 @@ export class CborSet { * Create a CborSet from any iterable of encodable items. Duplicates (by * canonical encoding) are removed. * - * NOTE: strings are iterable - `CborSet.from("abc")` is a THREE-element - * set of one-character strings, not a single-element set. Wrap in an - * array (`CborSet.from(["abc"])`) for the latter. + * Strings are iterable: `CborSet.from("abc")` is a three-element set of + * one-character strings. Use `CborSet.from(["abc"])` for a single element. */ static from(items: Iterable): CborSet { const set = new CborSet(); @@ -194,10 +193,8 @@ export class CborSet { } /** - * Iterate the stored `Cbor` elements lazily in canonical order. - * - * NOTE: this yields the stored `Cbor` nodes. Use `toArray()` for an eager - * array of extracted native values. + * Iterate the stored `Cbor` elements lazily in canonical order. Use + * `toArray()` for an eager array of extracted native values. */ *values(): Generator { yield* this; diff --git a/src/simple.ts b/src/simple.ts index d8cf209..aba8add 100644 --- a/src/simple.ts +++ b/src/simple.ts @@ -6,7 +6,7 @@ import { MajorType } from "./cbor-types"; import { encodeVarInt } from "./varint"; -import { f64CborData } from "./float"; +import { f64CborData, floatDisplayString } from "./float"; /** * Represents CBOR simple values (major type 7). @@ -34,7 +34,9 @@ export type Simple = * Returns the standard name of the simple value as a string. * * For `False`, `True`, and `Null`, this returns their lowercase string - * representation. For `Float` values, it returns their numeric representation. + * representation. For `Float` values, it returns Rust's `{:?}` rendering: + * `NaN`, `inf`, `-inf`, or the shortest round-trip decimal with at least one + * fractional digit (`42.0`, `1.5`, `1e21`). */ export const simpleName = (simple: Simple): string => { switch (simple.type) { @@ -49,9 +51,9 @@ export const simpleName = (simple: Simple): string => { if (Number.isNaN(v)) { return "NaN"; } else if (!Number.isFinite(v)) { - return v > 0 ? "Infinity" : "-Infinity"; + return v > 0 ? "inf" : "-inf"; } else { - return String(v); + return floatDisplayString(v); } } } @@ -77,8 +79,8 @@ export const isCborNaN = (simple: Simple): boolean => * - `False` encodes as `0xf4` * - `True` encodes as `0xf5` * - `Null` encodes as `0xf6` - * - `Float` values encode according to the IEEE 754 floating point rules, - * using the shortest representation that preserves precision. + * - `Float` values reduce to an integer when whole, otherwise encode in the + * shortest IEEE 754 width that preserves the value. */ export const simpleCborData = (simple: Simple): Uint8Array => { switch (simple.type) { @@ -122,11 +124,9 @@ export const simpleEquals = (a: Simple, b: Simple): boolean => { /** * Hash a Simple value. * - * This is a fast non-cryptographic hash (FNV-1a) used solely to drive - * in-process hash tables and dedup. It is not part of the deterministic - * CBOR wire format, which compares the encoded bytes, and it is not stable - * across processes or implementations. Do not persist these hash values or - * compare them externally. + * A non-cryptographic FNV-1a hash over the variant and float bits. It is + * not part of the wire format or of CBOR equality, and it is not stable + * across implementations; do not persist it. */ export const simpleHash = (simple: Simple): number => { // FNV-1a hash. diff --git a/src/sortable.ts b/src/sortable.ts index 5ae6aa3..a3fe4db 100644 --- a/src/sortable.ts +++ b/src/sortable.ts @@ -31,8 +31,8 @@ export function sortArrayByCborEncoding(array: readonly T[] } /** - * Sortable-by-CBOR-encoding interface shape. The `arraySortable` / - * `setSortable` helpers wrap any iterable into a `CBORSortable` view. + * A collection that can be sorted by CBOR encoding. `arraySortable` and + * `setSortable` wrap an array or a set in this interface. */ export interface CBORSortable { sortByCborEncoding(): T[]; diff --git a/src/sorted-byte-map.ts b/src/sorted-byte-map.ts index f501ad4..79d52c6 100644 --- a/src/sorted-byte-map.ts +++ b/src/sorted-byte-map.ts @@ -87,6 +87,20 @@ export class SortedByteMap { return n > 0 ? this.items[n - 1].key : undefined; } + /** + * The key at position `i` in ascending key order. Positional access lets + * two maps be walked in lockstep, and a single map be encoded, without + * materializing an entries array; the caller keeps `i` within `[0, size)`. + */ + keyAt(i: number): Uint8Array { + return this.items[i].key; + } + + /** The value at position `i` in ascending key order (see {@link keyAt}). */ + valueAt(i: number): V { + return this.items[i].value; + } + /** Map over each value (with its key) in ascending key order. */ map(fn: (value: V, key: Uint8Array) => T): T[] { return this.items.map((e) => fn(e.value, e.key)); diff --git a/src/string-util.ts b/src/string-util.ts index 306cbac..f75d4ce 100644 --- a/src/string-util.ts +++ b/src/string-util.ts @@ -1,5 +1,5 @@ /** - * String utilities for dCBOR, including Unicode normalization. + * String helpers for the diagnostic and hex-dump formatters. * * @module string-util */ @@ -15,21 +15,21 @@ export const flanked = (s: string, left: string, right: string): string => left + s + right; /** - * Check if a character is printable. Internal helper for {@link sanitized}. + * Check if a code point is printable (the reference's `is_printable`: any + * non-ASCII character, or ASCII 32-126). Internal helper for + * {@link sanitized}, which iterates code points, so an astral character (two + * UTF-16 code units) is one printable character here. * - * @param c - Character to check + * @param c - One code point, as a string * @returns True if printable */ const isPrintable = (c: string): boolean => { - if (c.length !== 1) return false; - const code = c.charCodeAt(0); - // Non-ASCII or ASCII printable (32-126) - return code > 127 || (code >= 32 && code <= 126); + const cp = c.codePointAt(0) ?? 0; + return cp > 127 || (cp >= 32 && cp <= 126); }; /** * Sanitize a string by replacing non-printable characters with dots. - * Returns None if the string has no printable characters. * * @param str - String to sanitize * @returns Sanitized string or undefined if no printable characters diff --git a/src/tag.ts b/src/tag.ts index 49fa261..cf906b2 100644 --- a/src/tag.ts +++ b/src/tag.ts @@ -58,6 +58,9 @@ export const Tag = { /** * Create a Tag from its numeric value, optionally with a name. * + * The returned object is frozen: a `Tag` is a value, as in the reference, + * and a store keeps the tags it is given by identity when they are frozen. + * * ```typescript * Tag.from(1, "date"); * Tag.from(12345); @@ -65,9 +68,9 @@ export const Tag = { */ from(value: TagValue, name?: string): Tag { if (name !== undefined) { - return { value, name }; + return Object.freeze({ value, name }); } - return { value }; + return Object.freeze({ value }); }, /** diff --git a/src/tags-store.ts b/src/tags-store.ts index 39286ed..43381fa 100644 --- a/src/tags-store.ts +++ b/src/tags-store.ts @@ -9,7 +9,7 @@ import { type Cbor } from "./cbor"; import { type CborNumber } from "./cbor-types"; -import type { Tag } from "./tag"; +import { Tag } from "./tag"; import { CborError } from "./error"; /** @@ -122,8 +122,14 @@ export class TagsStore implements ReadonlyTagsStore { * - Throws if a tag with the same value exists with a different name * - Allows re-registering the same tag value with the same name * + * The store holds frozen tags, as the reference stores clones it owns: a + * frozen argument (every `Tag.from` result) is kept by identity, an + * unfrozen object literal is copied, so later mutation of the caller's + * object never changes a lookup. + * * @param tag - The tag to register (must have a non-empty name) - * @throws Error if tag has no name, empty name, or conflicts with existing registration + * @throws {CborError} `Custom` if the tag has no name, an empty name, or + * conflicts with an existing registration * * @example * ```typescript @@ -149,20 +155,39 @@ export class TagsStore implements ReadonlyTagsStore { ); } - this._tagsByValue.set(key, tag); - this._tagsByName.set(name, tag); + const stored = Object.isFrozen(tag) ? tag : Tag.from(tag.value, name); + this._tagsByValue.set(key, stored); + this._tagsByName.set(name, stored); } /** * Register multiple tags; the conflict-throwing validation in `register()` - * applies per tag. + * applies per tag. Accepts any iterable, including a `readonly` array. */ - registerAll(tags: Tag[]): void { + registerAll(tags: Iterable): void { for (const tag of tags) { this.register(tag); } } + /** + * An independent copy of this store (the reference's `#[derive(Clone)]` + * on `TagsStore`). + * + * The clone holds the same frozen tags by identity and shares the + * summarizer functions, as the reference's `Arc` summarizers are shared. + * Registering a tag or setting a summarizer on either store leaves the + * other unchanged. The clone is a plain store; it never replaces the + * global store. + */ + clone(): TagsStore { + const copy = new TagsStore(); + for (const [key, tag] of this._tagsByValue) copy._tagsByValue.set(key, tag); + for (const [name, tag] of this._tagsByName) copy._tagsByName.set(name, tag); + for (const [key, summarizer] of this._summarizers) copy._summarizers.set(key, summarizer); + return copy; + } + /** * Register a custom summarizer function for a tag. * @@ -171,10 +196,10 @@ export class TagsStore implements ReadonlyTagsStore { * * @example * ```typescript - * store.setSummarizer(1, (cbor, flat) => { - * // Custom date formatting - * return `Date(${extractCbor(cbor)})`; - * }); + * store.setSummarizer(1, (cbor, flat) => ({ + * ok: true, + * value: `Date(${extractCbor(cbor)})`, + * })); * ``` */ setSummarizer(tagValue: CborNumber, summarizer: CborSummarizer): void { @@ -211,12 +236,7 @@ export class TagsStore implements ReadonlyTagsStore { return this._summarizers.get(key); } - /** - * Create a string key for a numeric tag value. - * Handles both number and bigint types. - * - * @private - */ + /** Map key for a tag value, equal for a `number` and the same `bigint`. */ private _valueKey(value: CborNumber): string { return value.toString(); } @@ -227,14 +247,24 @@ export class TagsStore implements ReadonlyTagsStore { // ============================================================================ /** - * Global singleton instance of the tags store. + * The slot the global store lives in. It is keyed on `globalThis` by a + * registered symbol rather than held in a module variable so that every copy + * of this module in a process - the ESM and CommonJS builds, or two bundled + * copies - resolves the SAME store, the way the reference's `GLOBAL_TAGS` + * static is one per process. The `@1` names the store's major version; bump + * it on a breaking `TagsStore` change so incompatible copies do not share. */ -let globalTagsStore: TagsStore | undefined; +const GLOBAL_TAGS_KEY = Symbol.for("@blockchaincommons/dcbor/global-tags-store@1"); + +interface GlobalSlot { + [GLOBAL_TAGS_KEY]?: TagsStore; +} /** * Get the global tags store instance. * - * Creates the instance on first access. + * Creates the instance on first access. One store per process for dcbor + * 1.x, shared by the ESM and CommonJS builds (see `GLOBAL_TAGS_KEY`). * * @returns The global TagsStore instance * @@ -244,10 +274,8 @@ let globalTagsStore: TagsStore | undefined; * store.register(Tag.from(999, 'myTag')); * ``` */ -export const getGlobalTagsStore = (): TagsStore => { - globalTagsStore ??= new TagsStore(); - return globalTagsStore; -}; +export const getGlobalTagsStore = (): TagsStore => + ((globalThis as GlobalSlot)[GLOBAL_TAGS_KEY] ??= new TagsStore()); /** * Execute a function with access to the global tags store. diff --git a/src/tags.ts b/src/tags.ts index 516ae53..a6356f6 100644 --- a/src/tags.ts +++ b/src/tags.ts @@ -1,9 +1,7 @@ /** - * Standard CBOR tag definitions from the IANA registry. - * - * This module defines the tag constants the library itself needs - most - * importantly the date and bignum tags - along with helpers to register them - * in a tags store and to resolve tag values to {@link Tag} objects. + * The tags the library itself defines - date (1) and the bignums (2, 3), as + * the reference's `tags.rs` - with helpers to register them in a tags store + * and to resolve tag values to {@link Tag} objects. * * @module tags * @see https://www.iana.org/assignments/cbor-tags/cbor-tags.xhtml @@ -11,36 +9,13 @@ import { Tag } from "./tag"; -// ============================================================================ -// Standard Date/Time Tags -// ============================================================================ - -/** - * Tag 0: Standard date/time string (RFC 3339) - */ -export const TAG_DATE_TIME_STRING = 0; - -/** - * Tag 1: Epoch-based date/time (seconds since 1970-01-01T00:00:00Z) - */ -export const TAG_EPOCH_DATE_TIME = 1; - -/** - * Tag 100: Epoch-based date (days since 1970-01-01) - */ -export const TAG_EPOCH_DATE = 100; - -// ============================================================================ -// Numeric Tags -// ============================================================================ - /** * Tag 2: Positive bignum (unsigned arbitrary-precision integer) */ export const TAG_POSITIVE_BIGNUM = 2; /** - * Tag 3: Negative bignum (signed arbitrary-precision integer) + * Tag 3: Negative bignum (arbitrary-precision negative integer) */ export const TAG_NEGATIVE_BIGNUM = 3; @@ -54,102 +29,6 @@ export const TAG_NAME_POSITIVE_BIGNUM = "positive-bignum"; */ export const TAG_NAME_NEGATIVE_BIGNUM = "negative-bignum"; -/** - * Tag 4: Decimal fraction [exponent, mantissa] - */ -export const TAG_DECIMAL_FRACTION = 4; - -/** - * Tag 5: Bigfloat [exponent, mantissa] - */ -export const TAG_BIGFLOAT = 5; - -// ============================================================================ -// Encoding Hints -// ============================================================================ - -/** - * Tag 21: Expected conversion to base64url encoding - */ -export const TAG_BASE64URL = 21; - -/** - * Tag 22: Expected conversion to base64 encoding - */ -export const TAG_BASE64 = 22; - -/** - * Tag 23: Expected conversion to base16 encoding - */ -export const TAG_BASE16 = 23; - -/** - * Tag 24: Encoded CBOR data item - */ -export const TAG_ENCODED_CBOR = 24; - -// ============================================================================ -// URI and Network Tags -// ============================================================================ - -/** - * Tag 32: URI (text string) - */ -export const TAG_URI = 32; - -/** - * Tag 33: base64url-encoded text - */ -export const TAG_BASE64URL_TEXT = 33; - -/** - * Tag 34: base64-encoded text - */ -export const TAG_BASE64_TEXT = 34; - -/** - * Tag 35: Regular expression (PCRE/ECMA262) - */ -export const TAG_REGEXP = 35; - -/** - * Tag 36: MIME message - */ -export const TAG_MIME_MESSAGE = 36; - -/** - * Tag 37: Binary UUID - */ -export const TAG_UUID = 37; - -// ============================================================================ -// Reference / UUID / Set Tags -// ============================================================================ - -/** - * Tag 256: string reference (namespace) - */ -export const TAG_STRING_REF_NAMESPACE = 256; - -/** - * Tag 257: binary UUID reference - */ -export const TAG_BINARY_UUID = 257; - -/** - * Tag 258: Set of values (array with no duplicates) - */ -export const TAG_SET = 258; - -// ============================================================================ -// Self-describing CBOR -// ============================================================================ - -/** - * Tag 55799: Self-describe CBOR (magic number 0xd9d9f7) - */ -export const TAG_SELF_DESCRIBE_CBOR = 55799; - // ============================================================================ // Global Tags Store Registration // ============================================================================ @@ -161,19 +40,16 @@ import { CborError } from "./error"; import { type Cbor } from "./cbor"; import { biguintFromUntaggedCbor, bigintFromNegativeUntaggedCbor } from "./bignum"; +/** + * Tag 1: Epoch-based date/time (seconds since 1970-01-01T00:00:00Z) + */ export const TAG_DATE = 1; -export const TAG_NAME_DATE = "date"; /** - * Register the standard tags (date, bignums) and their summarizers into - * `store`. - * - * Idempotent: tags already registered under the same name are skipped; - * registering a value under a DIFFERENT name still throws via the store's - * conflict validation. - * - * @param store - Target store; defaults to the global tags store. + * Name for tag 1 (date). */ +export const TAG_NAME_DATE = "date"; + /** Options for {@link registerStandardTags}. */ export interface RegisterStandardTagsOptions { /** @@ -186,16 +62,25 @@ export interface RegisterStandardTagsOptions { readonly bignum?: boolean | undefined; } +/** + * Register the standard tags (date, and the bignums with `bignum`) and their + * summarizers into `store`. + * + * Re-registering is idempotent and moves each standard name back to its + * standard value, as the reference's `insert_all` does: a store that had + * named tag 99 `date` names tag 1 `date` afterwards. Registering tag 1 (or + * 2/3 with `bignum`) under a different name throws `CborError` `Custom` + * from the store's conflict validation, before any summarizer is set. + * + * @param store - Target store; defaults to the global tags store. + */ export const registerStandardTags = ( store: TagsStore = getGlobalTagsStore(), options: RegisterStandardTagsOptions = {}, ): void => { const bignum = options.bignum ?? false; const tagsStore = store; - const dateTag = Tag.from(TAG_DATE, TAG_NAME_DATE); - if (tagsStore.tagForValue(TAG_DATE)?.name !== TAG_NAME_DATE) { - tagsStore.register(dateTag); - } + tagsStore.registerAll([Tag.from(TAG_DATE, TAG_NAME_DATE)]); // Set summarizer for date tag tagsStore.setSummarizer(TAG_DATE, (untaggedCbor: Cbor, _flat: boolean): SummarizerResult => { @@ -210,13 +95,10 @@ export const registerStandardTags = ( if (!bignum) return; // Register bignum tags (the reference's `num-bigint` build). - const biguintTag = Tag.from(TAG_POSITIVE_BIGNUM, TAG_NAME_POSITIVE_BIGNUM); - const bigintTag = Tag.from(TAG_NEGATIVE_BIGNUM, TAG_NAME_NEGATIVE_BIGNUM); - for (const tag of [biguintTag, bigintTag]) { - if (tagsStore.tagForValue(tag.value)?.name !== tag.name) { - tagsStore.register(tag); - } - } + tagsStore.registerAll([ + Tag.from(TAG_POSITIVE_BIGNUM, TAG_NAME_POSITIVE_BIGNUM), + Tag.from(TAG_NEGATIVE_BIGNUM, TAG_NAME_NEGATIVE_BIGNUM), + ]); // Summarizer for tag 2 (positive bignum) tagsStore.setSummarizer( @@ -248,31 +130,15 @@ export const registerStandardTags = ( }; /** - * Converts an array of tag values to their corresponding Tag objects. - * - * This function looks up each tag value in the global tag registry and returns - * an array of complete Tag objects. For any tag values that aren't - * registered in the global registry, it creates a basic Tag with just the - * value (no name). - * - * @param values - Array of numeric tag values to convert - * @returns Array of Tag objects corresponding to the input values + * Resolve tag values through the global tags store. A value the store does + * not know becomes an unnamed `Tag`. * * @example * ```typescript - * // Register some tags first * registerStandardTags(); - * - * // Convert tag values to Tag objects - * const tags = tagsForValues([1, 42, 999]); - * - * // The first tag (value 1) should be registered as "date" - * console.log(tags[0].value); // 1 - * console.log(tags[0].name); // "date" - * - * // Unregistered tags will have a value but no name - * console.log(tags[1].value); // 42 - * console.log(tags[2].value); // 999 + * const tags = tagsForValues([1, 42]); + * tags[0].name; // "date" + * tags[1].name; // undefined * ``` */ export const tagsForValues = (values: (number | bigint)[]): Tag[] => { @@ -282,7 +148,6 @@ export const tagsForValues = (values: (number | bigint)[]): Tag[] => { if (tag !== undefined) { return tag; } - // Create basic tag with just the value return Tag.from(value); }); }; diff --git a/src/utf8.ts b/src/utf8.ts new file mode 100644 index 0000000..542745b --- /dev/null +++ b/src/utf8.ts @@ -0,0 +1,121 @@ +/** + * UTF-8 validation failure description, mirroring `core::str::Utf8Error`. + * + * The decoder rejects malformed text with the WHATWG `TextDecoder` (fatal + * mode), whose error text is host-defined. To report the same message as the + * reference (`str::from_utf8` → `Utf8Error` → `Display`), the failing bytes + * are re-scanned here with a port of `core::str::validations:: + * run_utf8_validation`, which yields the reference's `(valid_up_to, + * error_len)` pair. + * + * @module utf8 + * @internal + */ + +/** + * `core::str::validations::utf8_char_width`: the sequence length a lead byte + * announces, or 0 for a byte that can never start a sequence (a continuation + * byte `80-bf`, the overlong leads `c0`/`c1`, or `f5-ff`). + */ +const utf8CharWidth = (lead: number): number => { + if (lead < 0x80) return 1; + if (lead < 0xc2) return 0; + if (lead < 0xe0) return 2; + if (lead < 0xf0) return 3; + if (lead < 0xf5) return 4; + return 0; +}; + +/** A byte that is not a UTF-8 continuation byte (`80-bf`). */ +const isNotContinuation = (byte: number): boolean => byte < 0x80 || byte > 0xbf; + +/** The reference's `(valid_up_to, error_len)`; `errorLength` is `undefined` for `None`. */ +export interface Utf8ErrorInfo { + readonly validUpTo: number; + readonly errorLength: number | undefined; +} + +/** + * Locate the first UTF-8 error in `bytes` the way `run_utf8_validation` + * does, or return `undefined` when the bytes are valid. + * + * `validUpTo` is the index of the offending lead byte. `errorLength` is the + * number of bytes to skip (1, 2 or 3) when an invalid byte is present, or + * `undefined` when the input ends inside a sequence. A present invalid byte + * always beats "incomplete": the continuation bytes are checked one at a + * time as they are read. + */ +export const findUtf8Error = (bytes: Uint8Array): Utf8ErrorInfo | undefined => { + const len = bytes.length; + let index = 0; + while (index < len) { + const first = bytes[index]; + if (first < 0x80) { + index++; + continue; + } + const start = index; + // The k-th continuation byte, or `undefined` when the input ends first. + const next = (): number | undefined => { + index++; + return index < len ? bytes[index] : undefined; + }; + const err = (errorLength: number | undefined): Utf8ErrorInfo => ({ + validUpTo: start, + errorLength, + }); + const width = utf8CharWidth(first); + if (width === 2) { + const b1 = next(); + if (b1 === undefined) return err(undefined); + if (isNotContinuation(b1)) return err(1); + } else if (width === 3) { + const b1 = next(); + if (b1 === undefined) return err(undefined); + const secondOk = + (first === 0xe0 && b1 >= 0xa0 && b1 <= 0xbf) || + (first >= 0xe1 && first <= 0xec && b1 >= 0x80 && b1 <= 0xbf) || + (first === 0xed && b1 >= 0x80 && b1 <= 0x9f) || + (first >= 0xee && first <= 0xef && b1 >= 0x80 && b1 <= 0xbf); + if (!secondOk) return err(1); + const b2 = next(); + if (b2 === undefined) return err(undefined); + if (isNotContinuation(b2)) return err(2); + } else if (width === 4) { + const b1 = next(); + if (b1 === undefined) return err(undefined); + const secondOk = + (first === 0xf0 && b1 >= 0x90 && b1 <= 0xbf) || + (first >= 0xf1 && first <= 0xf3 && b1 >= 0x80 && b1 <= 0xbf) || + (first === 0xf4 && b1 >= 0x80 && b1 <= 0x8f); + if (!secondOk) return err(1); + const b2 = next(); + if (b2 === undefined) return err(undefined); + if (isNotContinuation(b2)) return err(2); + const b3 = next(); + if (b3 === undefined) return err(undefined); + if (isNotContinuation(b3)) return err(3); + } else { + return err(1); + } + index++; + } + return undefined; +}; + +/** + * `Utf8Error`'s `Display` text for `bytes`, which must be invalid UTF-8: + * `invalid utf-8 sequence of N bytes from index I` or `incomplete utf-8 byte + * sequence from index I`. + */ +export const utf8ErrorDescription = (bytes: Uint8Array): string => { + const info = findUtf8Error(bytes); + if (info === undefined) { + // Unreachable when the decoder has already rejected the bytes; keep a + // truthful message rather than inventing an index. + return "invalid utf-8 sequence"; + } + return info.errorLength === undefined + ? `incomplete utf-8 byte sequence from index ${info.validUpTo}` + : `invalid utf-8 sequence of ${info.errorLength} bytes from index ${info.validUpTo}`; +}; diff --git a/src/varint.ts b/src/varint.ts index 0f3d5a8..cb33cf2 100644 --- a/src/varint.ts +++ b/src/varint.ts @@ -11,7 +11,7 @@ const typeBits = (t: MajorType): number => { /** * Write a CBOR head (major type + argument) straight into `writer`, avoiding * the intermediate `Uint8Array` that {@link encodeVarInt} allocates. This is - * the encoder hot path (every node emits a head). It MUST stay byte-identical + * the encoder hot path (every node emits a head). It must stay byte-identical * to {@link encodeVarInt}; the golden vectors cover both. */ export const writeVarInt = (writer: BufWriter, value: CborNumber, majorType: MajorType): void => { @@ -40,8 +40,8 @@ export const writeVarInt = (writer: BufWriter, value: CborNumber, majorType: Maj writer.writeBigUint64(BigInt(n)); } } else { - // See encodeVarInt: past MAX_SAFE_INTEGER the encoding collapses to the - // 9-byte u64 form (or OutOfRange above u64::MAX). BigInt keeps it lossless. + // Past MAX_SAFE_INTEGER the head is always the 9-byte u64 form + // (OutOfRange above u64::MAX). BigInt keeps the value lossless. const big = BigInt(value); if (big > U64_MAX) { throw CborError.outOfRange(); @@ -51,17 +51,20 @@ export const writeVarInt = (writer: BufWriter, value: CborNumber, majorType: Maj } }; +/** + * Encode a CBOR head (major type + argument) in its shortest form. + * + * @throws {CborError} `OutOfRange` for a negative, fractional, or + * above-u64 argument. + */ export const encodeVarInt = (value: CborNumber, majorType: MajorType): Uint8Array => { - // throw an error if the value is negative. if (value < 0) { throw CborError.outOfRange(); } - // throw an error if the value is a number with a fractional part. if (typeof value === "number" && hasFractionalPart(value)) { throw CborError.outOfRange(); } const type = typeBits(majorType); - // If the value is a `number` or a `bigint` that can be represented as a `number`, convert it to a `number`. if (isCborNumber(value) && value <= Number.MAX_SAFE_INTEGER) { value = Number(value); if (value <= 23) { @@ -84,7 +87,7 @@ export const encodeVarInt = (value: CborNumber, majorType: MajorType): Uint8Arra view.setUint32(1, value); return new Uint8Array(buffer); } else { - // Fits in MAX_SAFE_INTEGER + // Above u32, within the safe-integer range: UInt64 const buffer = new ArrayBuffer(9); const view = new DataView(buffer); view.setUint8(0, 0x1b | type); @@ -92,12 +95,9 @@ export const encodeVarInt = (value: CborNumber, majorType: MajorType): Uint8Arra return new Uint8Array(buffer); } } else { - // Bigint branch - value is strictly greater than `Number.MAX_SAFE_INTEGER`, - // therefore strictly greater than `0xffffffff`. The CBOR encoding rule - // collapses to: 9 bytes total (header `0x1b | type` + 8 big-endian bytes) - // if the value fits in u64, otherwise `OutOfRange`. We must NOT use - // `Math.log2(Number(value))` here - `Number(value)` is lossy past 2^53 - // and would mis-pick the encoding length for values near u64::MAX. + // Above MAX_SAFE_INTEGER (so above u32): the 9-byte u64 form if the value + // fits, otherwise OutOfRange. Compared as a bigint, since `Number(value)` + // is lossy here. const big = BigInt(value); if (big > U64_MAX) { throw CborError.outOfRange(); @@ -144,7 +144,7 @@ export const decodeVarIntData = ( } offset += 8; break; - default: // no additional info + default: // 0-23: the argument is the additional info itself value = additionalInfo; break; } diff --git a/src/walk.ts b/src/walk.ts index 900d0b6..894e70a 100644 --- a/src/walk.ts +++ b/src/walk.ts @@ -1,8 +1,5 @@ /** - * Tree traversal system for CBOR data structures. - * - * This module provides a visitor pattern implementation for traversing - * CBOR trees, allowing users to inspect and process elements at any depth. + * Depth-first traversal of a CBOR tree with a visitor function. * * @module walk */ @@ -74,10 +71,6 @@ export const edgeLabel = (edge: EdgeTypeVariant): string | undefined => { export type WalkElement = { type: "single"; cbor: Cbor } | { type: "keyvalue"; key: Cbor; value: Cbor }; -/** - * Helper functions for WalkElement - */ - /** * Returns the single CBOR element if this is a 'single' variant. * @@ -140,9 +133,8 @@ const cloneState = (s: S): S => { if (s === null) return s; const t = typeof s; if (t !== "object" && t !== "function") return s; - // `structuredClone` is a host-provided global available in modern Node - // (≥ 17) and every modern browser; declare it inline so eslint's - // `no-undef` is satisfied without a project-wide globals declaration. + // Reached through `globalThis` so lint needs no globals declaration for + // the host-provided `structuredClone`. return (globalThis as { structuredClone(v: unknown): unknown }).structuredClone(s) as S; }; diff --git a/tests/baseline/README.md b/tests/baseline/README.md index af317be..452940e 100644 --- a/tests/baseline/README.md +++ b/tests/baseline/README.md @@ -1,19 +1,18 @@ -# Frozen baseline build (P1.1 differential harness) +# Baseline build for the differential harness -`dcbor-baseline.mjs` is the dependency-free ESM bundle of @blockchaincommons/dcbor built from -commit `187769a273176a2c6089fb412871226b6db1e795` - the frozen pre-redesign -wire-format reference. (Its trailing `sourceMappingURL` comment was removed -because the `.map` file is not vendored.) +`dcbor-baseline.mjs` is the dependency-free ESM bundle of the library built +from commit `187769a273176a2c6089fb412871226b6db1e795`, before the public API +was renamed to its current form. (Its trailing `sourceMappingURL` comment was +removed because the `.map` file is not vendored.) The differential corpus harness (`tests/differential.test.ts`) encodes every -corpus input with BOTH this baseline and the working tree, asserting +corpus input with both this baseline and the working tree, asserting byte-identical output and identical decode outcomes/error codes. The harness also pins this file's sha256 (`BASELINE_SHA256`) so an accidental rebuild or copy mishap cannot silently turn the differential into a self-comparison. -Do NOT regenerate this file during the API redesign. It is only re-baselined -at P4.1 (proof re-baseline), by deliberate decision. To re-baseline, build the -RECORDED COMMIT (never HEAD) in a detached worktree: +Re-baseline only by deliberate decision. Build the recorded commit (never +HEAD) in a detached worktree: git worktree add /tmp/dcbor-baseline-build cd /tmp/dcbor-baseline-build && bun install && bun run build @@ -23,5 +22,9 @@ RECORDED COMMIT (never HEAD) in a detached worktree: # and the commit hash in this README - all in one reviewed diff. git worktree remove /tmp/dcbor-baseline-build +A new baseline with a different public API also needs its adapter in +`tests/vectors/recipes.ts` (`baselineAdapterFor`) and the type shim in +`dcbor-baseline.d.mts` updated. + Baseline commit: 187769a273176a2c6089fb412871226b6db1e795 Baseline sha256: ffb0bf6acdafaf01fbb6360497f96cb0f821d4d602f6f343cddd507a302fb72c diff --git a/tests/baseline/dcbor-baseline.d.mts b/tests/baseline/dcbor-baseline.d.mts index e2256d3..aacfcfa 100644 --- a/tests/baseline/dcbor-baseline.d.mts +++ b/tests/baseline/dcbor-baseline.d.mts @@ -1,9 +1,8 @@ /** - * Type shim for the frozen pre-redesign baseline bundle (see README.md). + * Type shim for the baseline bundle (see README.md). * - * Only the members the differential harness's adapter touches are declared. - * These signatures are FROZEN with the bundle: the baseline never changes - * (except at a deliberate P4.1 re-baseline), so this file never drifts. + * Only the members the differential harness's adapter and the hex property + * tests touch are declared. The signatures change only with the bundle. */ export declare function cborData(value: unknown): Uint8Array; diff --git a/tests/bignum.test.ts b/tests/bignum.test.ts index 9bd436e..5bcdb74 100644 --- a/tests/bignum.test.ts +++ b/tests/bignum.test.ts @@ -1,8 +1,6 @@ /** - * Tests for CBOR bignum (tags 2 and 3) support. - * - * This file is a complete 1:1 translation of Rust's tests/num_bigint.rs. - * All 57 unique test cases from the Rust version are translated here. + * Tests for CBOR bignum (tags 2 and 3) support, ported from Rust's + * tests/num_bigint.rs. */ import { describe, it, expect } from "vitest"; @@ -527,7 +525,7 @@ describe("Encoding length", () => { // c2 41 01 = tag 2, 1-byte bstr, 0x01 expect(one.toData().length).toBe(3); - // 2^64 should be tag + 9-byte length prefix + 9 bytes = 11 bytes + // 2^64: tag head + byte-string head + 9 content bytes = 11 bytes const big = biguintToCbor(1n << 64n); expect(big.toData().length).toBe(11); }); @@ -546,7 +544,6 @@ describe("Tag summarizer", () => { flat: true, }; - // Register tags before summarizer tests it("summarizer positive bignum", () => { registerStandardTags(undefined, { bignum: true }); const encoded = biguintToCbor(256n); diff --git a/tests/cli.test.ts b/tests/cli.test.ts index 69ae0a5..7b5dd70 100644 --- a/tests/cli.test.ts +++ b/tests/cli.test.ts @@ -1,18 +1,10 @@ /** - * CLI Tests - Comparison with bc-dcbor-cli + * Compares output with the Rust `dcbor` CLI + * (https://github.com/BlockchainCommons/bc-dcbor-cli), whose tests these are + * based on. * - * REQUIREMENTS: - * These tests require the bc-dcbor-cli tool to be installed on your system. - * The tests execute the actual dcbor CLI command to verify our TypeScript - * implementation produces identical output to the Rust reference implementation. - * - * Installation: - * cargo install bc-dcbor-cli - * - * To run these tests: - * npm run test-cli - * - * Based on tests from: https://github.com/BlockchainCommons/bc-dcbor-cli + * Requires the CLI on PATH (`cargo install bc-dcbor-cli`). The file is + * excluded from the default test run in vitest.config.ts. */ import { execSync } from "child_process"; @@ -79,7 +71,6 @@ function runDcborWithInput(args: string[], input: string): string { * Handles: numbers, strings, booleans, null, arrays, maps */ function parseDiagnostic(input: string): CborInput { - // Trim whitespace input = input.trim(); // Handle null @@ -208,8 +199,6 @@ function toHex(value: CborInput): string { */ function toDiagnostic(value: CborInput): string { const cborValue = cbor(value); - // P3.8: Cbor.toString() now returns `Cbor(0x…)`; flat diagnostic moved to - // diagnostic(c, { flat: true }) - identical string to the old toString(). return diagnostic(cborValue, { flat: true }); } @@ -225,16 +214,13 @@ function toAnnotatedHex(value: CborInput): string { * Convert hex string to diagnostic notation */ function hexToDiagnostic(hexStr: string): string { - // Remove any whitespace hexStr = hexStr.replace(/\s+/g, ""); - // Convert hex to bytes const bytes = new Uint8Array(hexStr.length / 2); for (let i = 0; i < hexStr.length; i += 2) { bytes[i / 2] = parseInt(hexStr.substr(i, 2), 16); } - // Decode and convert to diagnostic const decoded = decodeCbor(bytes); return diagnostic(decoded, { flat: true }); } diff --git a/tests/conveniences.test.ts b/tests/conveniences.test.ts index d9fa786..421a55e 100644 --- a/tests/conveniences.test.ts +++ b/tests/conveniences.test.ts @@ -1,6 +1,7 @@ /** - * Regression tests for the medium-severity parity fixes (M1, M4, M5) from the - * dCBOR Rust parity audit. + * Rust parity of the conveniences: tag comparison across number/bigint, + * tagged-content errors, fixed-width unsigned and float extraction, and + * summarizer error rendering. */ import { describe, test, expect } from "vitest"; @@ -10,25 +11,30 @@ import { hasTag, getTaggedContent, expectTaggedContent, + expectUnsigned, validateTag, asFloat, expectFloat, TagsStore, CborError, + Tag, + decodeCbor, + hexToBytes, + registerStandardTags, + getGlobalTagsStore, } from "../src"; import { diagnostic } from "../src/diag"; -describe("M1: value-normalized tag equality (number/bigint boundary)", () => { +describe("value-normalized tag equality (number/bigint boundary)", () => { test("hasTag matches across the number/bigint divide", () => { const tagNum = taggedValue(100, 1); // tag stored as number 100 const tagBig = taggedValue(100n, 1); // tag stored as bigint 100n - // Rust's Tag equality is value-only over u64; in JS `100n === 100` is false, - // so the raw === used previously would have rejected these. + // Rust's Tag equality is value-only over u64; in JS `100n === 100` is + // false, so the comparison normalizes the value first. expect(hasTag(tagNum, 100n)).toBe(true); expect(hasTag(tagBig, 100)).toBe(true); expect(hasTag(tagNum, 100)).toBe(true); expect(hasTag(tagBig, 100n)).toBe(true); - // Still rejects a genuinely different tag value. expect(hasTag(tagNum, 101)).toBe(false); expect(hasTag(tagNum, 101n)).toBe(false); }); @@ -43,7 +49,7 @@ describe("M1: value-normalized tag equality (number/bigint boundary)", () => { expect(expectTaggedContent(tagNum, 100n).type).toBeDefined(); }); - test("expectTaggedContent still throws WrongTag for a real mismatch", () => { + test("expectTaggedContent throws for a real mismatch", () => { const t = taggedValue(100, 1); expect(() => expectTaggedContent(t, 101)).toThrow(CborError); }); @@ -60,7 +66,115 @@ describe("M1: value-normalized tag equality (number/bigint boundary)", () => { }); }); -describe("M5: asFloat/expectFloat coerce integers (Rust TryFrom for f64)", () => { +describe("expectTaggedContent names both tags like try_into_expected_tagged_value", () => { + // Executed on dcbor 0.25.2. A numeric expected tag is `Tag::with_value` + // (unnamed, whatever the store knows); a `Tag` keeps its name; the actual + // tag is the one the node carries (decoded nodes carry none). + const message = (f: () => unknown): string => { + try { + f(); + } catch (e) { + return e instanceof Error ? e.message : String(e); + } + return "(no throw)"; + }; + const node = () => taggedValue(Tag.from(99, "x"), 0); + + test("numeric expected tags stay unnamed, before and after registration", () => { + expect(message(() => expectTaggedContent(node(), 40000))).toBe( + "expected CBOR tag 40000, but got x", + ); + registerStandardTags(); + getGlobalTagsStore().register(Tag.from(40000, "kv")); + expect(message(() => expectTaggedContent(node(), 40000))).toBe( + "expected CBOR tag 40000, but got x", + ); + expect(message(() => expectTaggedContent(node(), 40000n))).toBe( + "expected CBOR tag 40000, but got x", + ); + }); + + test("a Tag argument keeps its name", () => { + expect(message(() => expectTaggedContent(node(), Tag.from(40000, "kvexp")))).toBe( + "expected CBOR tag kvexp, but got x", + ); + expect(expectTaggedContent(node(), Tag.from(99, "anything")).type).toBe(0); + }); + + test("a decoded node reports its bare number", () => { + expect(message(() => expectTaggedContent(decodeCbor(hexToBytes("d86300")), 40000))).toBe( + "expected CBOR tag 40000, but got 99", + ); + }); +}); + +describe("expectUnsigned with a width wraps negatives like u*::try_from", () => { + // Executed on dcbor 0.25.2 (`u8/u32/u64::try_from(CBOR)`): the magnitude on + // the wire must fit the width (else OutOfRange); a negative then wraps. + const code = (f: () => unknown): string => { + try { + f(); + } catch (e) { + return CborError.isCborError(e) ? e.code : "foreign"; + } + return "(no throw)"; + }; + const wrap = (v: number | bigint, width: 8 | 16 | 32 | 64) => + expectUnsigned(cbor(v), { width, wrapNegative: true }); + + test("width 8", () => { + expect(wrap(-1, 8)).toBe(255); + expect(wrap(-256, 8)).toBe(0); + expect(code(() => wrap(-257, 8))).toBe("OutOfRange"); + expect(code(() => wrap(256, 8))).toBe("OutOfRange"); + expect(wrap(255, 8)).toBe(255); + expect(wrap(0, 8)).toBe(0); + }); + + test("width 16 and 32", () => { + expect(wrap(-1, 16)).toBe(65535); + expect(code(() => wrap(65536, 16))).toBe("OutOfRange"); + expect(wrap(-1, 32)).toBe(4294967295); + expect(wrap(-(2 ** 32), 32)).toBe(0); + expect(code(() => wrap(-(2 ** 32) - 1, 32))).toBe("OutOfRange"); + expect(code(() => wrap(2 ** 32, 32))).toBe("OutOfRange"); + }); + + test("width 64 returns bigint above the safe range", () => { + expect(wrap(-(2n ** 64n), 64)).toBe(0); + expect(wrap(-1, 64)).toBe(18446744073709551615n); + expect(wrap(2n ** 64n - 1n, 64)).toBe(18446744073709551615n); + expect(wrap(2n ** 53n, 64)).toBe(9007199254740992n); + expect(wrap(5, 64)).toBe(5); + // −2^64 is the 65-bit negative `3bffffffffffffffff`, decodable but not a JS number. + expect( + expectUnsigned(decodeCbor(hexToBytes("3bffffffffffffffff")), { + width: 64, + wrapNegative: true, + }), + ).toBe(0); + }); + + test("without wrapNegative a negative is WrongType; non-integers are WrongType", () => { + expect(code(() => expectUnsigned(cbor(-1), { width: 8 }))).toBe("WrongType"); + expect(code(() => expectUnsigned(cbor(1.5), { width: 8, wrapNegative: true }))).toBe( + "WrongType", + ); + expect(code(() => expectUnsigned(cbor("1"), { width: 8, wrapNegative: true }))).toBe( + "WrongType", + ); + expect(code(() => expectUnsigned(cbor(300), { width: 8 }))).toBe("OutOfRange"); + expect(expectUnsigned(cbor(300), { width: 16 })).toBe(300); + }); + + test("without options there is no width bound and a negative is WrongType", () => { + expect(expectUnsigned(cbor(300))).toBe(300); + expect(expectUnsigned(cbor(2n ** 64n - 1n))).toBe(18446744073709551615n); + expect(code(() => expectUnsigned(cbor(-1)))).toBe("WrongType"); + }); +}); + +describe("asFloat/expectFloat coerce integers (Rust TryFrom for f64)", () => { test("asFloat coerces Unsigned/Negative and passes through floats", () => { expect(asFloat(cbor(42))).toBe(42); expect(asFloat(cbor(-5))).toBe(-5); @@ -96,15 +210,13 @@ describe("M5: asFloat/expectFloat coerce integers (Rust TryFrom for f64)", }); }); -describe("M4: summarizer error rendered via the full Error Display", () => { +describe("summarizer error rendered via the full Error Display", () => { test("non-Custom/non-WrongTag summarizer errors show the Rust message", () => { const store = new TagsStore(); store.register({ value: 1234, name: "thing" }); - // Summarizer that always fails with WrongType. store.setSummarizer(1234, () => ({ ok: false, error: CborError.wrongType() })); const tagged = taggedValue(1234, 1); const out = diagnostic(tagged, { summarize: true, tags: store }); - // Previously this rendered the bare variant id ``. expect(out).toBe(""); }); diff --git a/tests/corpus/corpus.ts b/tests/corpus/corpus.ts index 234bbbc..c5e7295 100644 Binary files a/tests/corpus/corpus.ts and b/tests/corpus/corpus.ts differ diff --git a/tests/date.test.ts b/tests/date.test.ts index 194c5d2..99267bb 100644 --- a/tests/date.test.ts +++ b/tests/date.test.ts @@ -1,16 +1,29 @@ /** - * Regression tests for M2 - strict `CborDate.fromString` parsing. - * - * Mirrors Rust `Date::from_string`: accept only strict RFC-3339 date-times - * (seconds + explicit Z/offset) or bare `YYYY-MM-DD` dates (UTC midnight); - * reject everything else, including the lenient/engine-dependent forms the old - * `new Date(value)` accepted. + * `CborDate` tests against Rust's `Date`: string parsing, the representable + * range, component constructors, the (seconds, nanoseconds) model, `WrongTag` + * naming and display. */ import { describe, test, expect } from "vitest"; -import { CborDate, CborError, decodeCbor, encodeCbor, bytesToHex } from "../src"; +import { + CborDate, + CborError, + decodeCbor, + decodeWith, + encodeCbor, + bytesToHex, + hexToBytes, + taggedValue, + Tag, + registerStandardTags, + getGlobalTagsStore, + tagsForValues, + cborEquals, +} from "../src"; +import { diagnostic } from "../src/diag"; +import { hexAnnotated } from "../src/dump"; -describe("M2: strict CborDate.fromString", () => { +describe("strict CborDate.fromString", () => { test("accepts strict RFC-3339 date-times", () => { expect(() => CborDate.fromString("2023-02-08T15:30:45Z")).not.toThrow(); expect(() => CborDate.fromString("2023-02-08T15:30:45.5Z")).not.toThrow(); @@ -57,12 +70,12 @@ describe("M2: strict CborDate.fromString", () => { test("round-trips a whole-second timestamp through encode", () => { const d = CborDate.fromString("2022-03-21T18:24:31Z"); - // Same instant as Rust's encode_date-style vector. + // The instant of Rust's `format_date` vector. expect(d.epochSeconds).toBe(1647887071); }); }); -describe("range: the reference's representable timestamps (review N2)", () => { +describe("range: the reference's representable timestamps", () => { // chrono's NaiveDateTime::MIN / MAX as Unix seconds (executed on dcbor 0.25.2). const MIN = -8334601228800; const MAX = 8210266876799; @@ -241,6 +254,174 @@ describe("fromDate follows `from_datetime`: exact milliseconds, chrono's range", }); }); +describe("the (seconds, nanoseconds) model: chrono's instant, not one f64", () => { + // Every expected string/byte sequence below was executed on dcbor 0.25.2 + // (`Date::from_string` / `from_timestamp`, then `to_string()` and + // `to_cbor_data()`). + const hexOf = (d: CborDate): string => bytesToHex(encodeCbor(d.toCbor())); + const MIN = -8334601228800; + + test("a leap second displays as :60 and keeps its wire value", () => { + const leap = CborDate.fromString("2023-12-25T10:30:60Z"); + expect(leap.toString()).toBe("2023-12-25T10:30:60Z"); + expect(leap.epochSeconds).toBe(1703500260); + expect(hexOf(leap)).toBe("c11a658959e4"); + expect(CborDate.fromString("2023-12-25T23:59:60.5+01:00").toString()).toBe( + "2023-12-25T22:59:60Z", + ); + expect(CborDate.fromString("2023-12-31T23:59:60Z").toString()).toBe("2023-12-31T23:59:60Z"); + // After a CBOR round trip the leap second is an ordinary instant again. + expect(CborDate.fromTaggedCbor(decodeCbor(encodeCbor(leap.toCbor()))).toString()).toBe( + "2023-12-25T10:31:00Z", + ); + }); + + test("equality and ordering compare the pair, as chrono does", () => { + const leap = CborDate.fromString("2023-12-25T10:30:60Z"); + const next = CborDate.fromEpochSeconds(1703500260); + const before = CborDate.fromString("2023-12-25T10:30:59.999999999Z"); + expect(leap.epochSeconds).toBe(next.epochSeconds); + expect(leap.equals(next)).toBe(false); + expect(leap.compare(next)).toBe(-1); + expect(next.compare(leap)).toBe(1); + expect(before.compare(leap)).toBe(-1); + expect(leap.equals(CborDate.fromString("2023-12-25T10:30:60Z"))).toBe(true); + expect(CborDate.fromEpochSeconds(1.5).equals(CborDate.fromEpochSeconds(1.5))).toBe(true); + expect(CborDate.fromEpochSeconds(1.5).compare(CborDate.fromEpochSeconds(1.25))).toBe(1); + }); + + test("sub-second rounding of the f64 wire value does not move the displayed second", () => { + // 45.999999999 s: the f64 sum rounds up to …46.0 (that is the wire + // value, on both sides), but the instant is still second 45. + const d = CborDate.fromString("2023-12-25T10:30:45.999999999Z"); + expect(d.toString()).toBe("2023-12-25T10:30:45Z"); + expect(d.epochSeconds).toBe(1703500246); + expect(hexOf(d)).toBe("c11a658959d6"); + }); + + test("the range check applies to the truncated whole seconds", () => { + const d = CborDate.fromEpochSeconds(MIN - 0.5); + expect(hexOf(d)).toBe("c13b000007948cf211ff"); + expect(d.toString()).toBe("-262143-01-01"); + expect(CborDate.fromEpochSeconds(MIN - 0.999).toString()).toBe("-262143-01-01"); + expect(() => CborDate.fromEpochSeconds(MIN - 1)).toThrow( + "timestamp outside the representable range", + ); + // MIN - 0.5 as a tag-1 float decodes to MIN as well. + const belowMin = decodeCbor(encodeCbor(taggedValue(1, MIN - 0.5))); + expect(CborDate.fromTaggedCbor(belowMin).toString()).toBe("-262143-01-01"); + expect(hexOf(CborDate.fromTaggedCbor(belowMin))).toBe("c13b000007948cf211ff"); + }); + + test("fromDate and toDate carry the millisecond part exactly", () => { + const d = CborDate.fromDate(new Date(4190400121)); + expect(d.toDate().getTime()).toBe(4190400121); + expect(CborDate.fromDate(new Date(-500)).toDate().getTime()).toBe(-500); + expect(CborDate.fromDate(new Date(-500)).epochSeconds).toBe(-0.5); + expect(CborDate.fromDate(new Date(-500)).compare(CborDate.fromEpochSeconds(-1))).toBe(1); + }); +}); + +describe("NaN saturates to the epoch, ±Infinity is InvalidDate (from_timestamp parity)", () => { + const hexOf = (d: CborDate): string => bytesToHex(encodeCbor(d.toCbor())); + test("construction", () => { + const d = CborDate.fromEpochSeconds(NaN); + expect(d.toString()).toBe("1970-01-01"); + expect(d.epochSeconds).toBe(0); + expect(hexOf(d)).toBe("c100"); + expect(d.equals(CborDate.fromEpochSeconds(0))).toBe(true); + for (const s of [Infinity, -Infinity]) { + expect(() => CborDate.fromEpochSeconds(s)).toThrow("non-finite timestamp"); + } + expect(() => CborDate.fromEpochSeconds(Infinity)).toThrow(CborError); + }); + test("tag-1 decode", () => { + const dec = (hex: string) => CborDate.fromTaggedCbor(decodeCbor(hexToBytes(hex))); + expect(dec("c1f97e00").toString()).toBe("1970-01-01"); + expect(hexOf(dec("c1f97e00"))).toBe("c100"); + expect(() => dec("c1f97c00")).toThrow("non-finite timestamp"); + expect(() => dec("c1f9fc00")).toThrow("non-finite timestamp"); + }); + test("an invalid JS Date has no reference analog and is still rejected", () => { + expect(() => CborDate.fromDate(new Date(NaN))).toThrow("non-finite timestamp"); + }); +}); + +describe("WrongTag names the expected and actual tags as the reference does", () => { + // Executed on dcbor 0.25.2: `Date::from_tagged_cbor` reports + // `WrongTag(cbor_tags()[0], tag)`, where the expected tag's name comes from + // the global store (`tags_for_values`) and the actual tag keeps the name it + // was built with (a decoded tag has none). This file's global store starts + // empty; the registered rows run after `registerStandardTags()`. + const message = (f: () => unknown): string => { + try { + f(); + } catch (e) { + return e instanceof Error ? e.message : String(e); + } + return "(no throw)"; + }; + + test("before any registration the expected tag is bare `1`", () => { + expect(CborDate.fromEpochSeconds(0).cborTags()[0]?.name).toBeUndefined(); + expect(CborDate.codec.tags?.[0]?.name).toBeUndefined(); + expect(message(() => CborDate.fromTaggedCbor(taggedValue(2, 0)))).toBe( + "expected CBOR tag 1, but got 2", + ); + expect(message(() => CborDate.fromTaggedCbor(taggedValue(Tag.from(40000, "adhoc"), 0)))).toBe( + "expected CBOR tag 1, but got adhoc", + ); + expect(message(() => decodeWith(hexToBytes("d99c4000"), CborDate.codec))).toBe( + "expected CBOR tag 1, but got 40000", + ); + }); + + test("after registration the expected tag is `date`; the actual tag keeps its own name", () => { + registerStandardTags(); + getGlobalTagsStore().register(Tag.from(40000, "custom")); + expect(CborDate.fromEpochSeconds(0).cborTags()[0]?.name).toBe("date"); + expect(CborDate.codec.tags?.[0]?.name).toBe("date"); + const [custom] = tagsForValues([40000]); + expect(custom?.name).toBe("custom"); + expect(message(() => CborDate.fromTaggedCbor(taggedValue(custom ?? 40000, 0)))).toBe( + "expected CBOR tag date, but got custom", + ); + expect(message(() => CborDate.fromTaggedCbor(taggedValue(Tag.from(40000, "adhoc"), 0)))).toBe( + "expected CBOR tag date, but got adhoc", + ); + // A decoded node carries no name, even when the store knows one. + expect(message(() => CborDate.fromTaggedCbor(decodeCbor(hexToBytes("d99c4000"))))).toBe( + "expected CBOR tag date, but got 40000", + ); + expect(message(() => CborDate.fromTaggedCbor(taggedValue(2, 0)))).toBe( + "expected CBOR tag date, but got 2", + ); + const err = (() => { + try { + CborDate.fromTaggedCbor(taggedValue(Tag.from(40000, "adhoc"), 0)); + } catch (e) { + return e; + } + return undefined; + })(); + expect(CborError.isCborError(err) && err.code).toBe("WrongTag"); + }); + + test("the carried name never reaches the wire, the diagnostic or the hex dump", () => { + const named = taggedValue(Tag.from(40000, "adhoc"), 0); + const unnamed = taggedValue(40000, 0); + expect(encodeCbor(named)).toEqual(encodeCbor(unnamed)); + expect(diagnostic(named)).toBe(diagnostic(unnamed)); + expect(hexAnnotated(named)).toBe(hexAnnotated(unnamed)); + expect(hexAnnotated(named)).toContain("custom"); // the store's name, not the node's + expect(cborEquals(named, unnamed)).toBe(true); + const dateNode = CborDate.fromEpochSeconds(0).taggedCbor(); + expect(dateNode.type === 7 ? undefined : dateNode.type === 6 ? dateNode.tagName : "").toBe( + "date", + ); + }); +}); + describe("toString prints the reference's Display", () => { // `%Y-%m-%d` at 00:00:00 (a fraction does not count), else RFC 3339 to the // second; years outside 0–9999 carry a sign and at least four digits. diff --git a/tests/differential.test.ts b/tests/differential.test.ts index bd16f32..8f5f189 100644 --- a/tests/differential.test.ts +++ b/tests/differential.test.ts @@ -1,12 +1,11 @@ /** - * Differential corpus harness (API_REDESIGN_PLAN P1.1b) - the mechanism that - * proves the API redesign never changes the wire format. + * Differential corpus harness: the working tree against a baseline build. * * Every recipe in the ~93k combinatorial corpus (tests/corpus/corpus.ts) is - * materialized and encoded TWICE: once with the frozen baseline bundle + * materialized and encoded twice: once with the baseline bundle * (tests/baseline/dcbor-baseline.mjs, built from the commit recorded in * tests/baseline/README.md) and once with the working tree (../src). The - * encodings must be byte-identical - including which inputs throw, and with + * encodings must be byte-identical, including which inputs throw and with * which CborError code. * * Decode differential: on a deterministic stride, the encoded bytes plus @@ -14,31 +13,23 @@ * both builds; accept/reject outcome, re-encoded bytes, and error codes must * match exactly. The committed golden decode vectors also run through both. * - * ## Tombstones (P3.5 / P3.7) - * - * When the breaking wave lands, the two flagged input shapes STOP encoding - * and start throwing a directive error. Flip them here by adding the plan - * task id to EXPECTED_TOMBSTONES - the harness then asserts - * baseline-encodes → working-tree-throws for those categories, and byte - * equality for everything else. That flip is the ONLY allowed differential - * change during the redesign. - * - * ## When the public API is renamed (Phase 3) - * - * Write a new adapter for the redesigned surface in tests/vectors/recipes.ts - * and point `current` below at it. The baseline keeps using `adapterFor`. - * Recipes, corpus, and fixtures must NOT change. + * Deliberate behavior changes since the baseline are recognized by the + * `isKnown*Change` predicates below. Categories marked `tombstone` hold + * input shapes the baseline encoded and the working tree rejects; for those + * the harness asserts the directive error instead of byte equality. */ import * as baselineMod from "./baseline/dcbor-baseline.mjs"; import * as src from "../src"; import { - adapterFor, - redesignedAdapterFor, + baselineAdapterFor, + currentAdapterFor, bytesToHex, decodeOutcome, encodeOutcome, hexToBytes, + type Recipe, + type RemovedInputShape, type VectorApi, } from "./vectors/recipes"; import { categories, corpusSize } from "./corpus/corpus"; @@ -47,29 +38,24 @@ import { dirname, join } from "node:path"; import { fileURLToPath } from "node:url"; /** - * Plan tasks whose tombstone throw has LANDED in the working tree, mapped to - * the CborError code of the directive error they must throw. Pre-wave this - * is empty. When P3.5 lands add e.g. `["P3.5", "Custom"]`; same for P3.7. - * Do not add anything else, ever. (The directive error MUST be a CborError - - * a non-CborError throw crashes the category loudly by design.) + * The CborError code each removed input shape throws. The directive error + * must be a CborError: a non-CborError throw fails the category loudly. */ -const EXPECTED_TOMBSTONES = new Map<"P3.5" | "P3.7", string>([ - ["P3.5", "Custom"], - ["P3.7", "Custom"], +const EXPECTED_TOMBSTONES = new Map([ + ["tag-value-literal", "Custom"], + ["tagged-cbor-only", "Custom"], ]); type DecodeDiffOutcome = { ok: true; hex: string } | { ok: false; code: string; stage?: string }; /** - * Deliberate decode-behavior change landed AFTER the frozen baseline: - * byte-string/text lengths >= 2^53 (bigint after narrowing) now report - * `Underrun` like Rust's usize bounds check instead of the baseline's - * `OutOfRange` input guard - see RUST_DIVERGENCES.md. Both builds still - * reject; only the code differs. The baseline's decode stage raised - * OutOfRange ONLY from that bigint-length guard (asserted by the golden - * fixture hygiene test), so this exact code pair - and nothing else - is - * the signature of the change. Fold into the baseline at the next - * deliberate re-baseline. + * Decode-behavior change since the baseline: byte-string/text lengths >= 2^53 + * (bigint after narrowing) report `Underrun`, like the reference's + * `parse_bytes` bounds check, instead of the baseline's `OutOfRange` input + * guard. Both builds reject; only the code differs. The baseline's decode + * stage raised OutOfRange only from that bigint-length guard (asserted by the + * golden fixture hygiene test), so this exact code pair - and nothing else - + * is the signature of the change. */ const isKnownLengthCodeChange = (base: DecodeDiffOutcome, cur: DecodeDiffOutcome): boolean => !base.ok && @@ -80,16 +66,84 @@ const isKnownLengthCodeChange = (base: DecodeDiffOutcome, cur: DecodeDiffOutcome cur.code === "Underrun"; /** - * Integrity pin for the frozen baseline. Without this, an accidental - * re-baseline (or a build/copy mishap) silently turns the differential into - * a self-comparison that passes vacuously. Update ONLY at the deliberate - * P4.1 proof re-baseline, together with tests/baseline/README.md. + * Decode-behavior change since the baseline: a text string starting with + * U+FEFF keeps that code point (Rust `String::from_utf8` parity; the baseline + * used the WHATWG TextDecoder default, which strips a leading BOM, so its + * re-encode dropped the three bytes). Signature: both builds accept, the + * working tree re-encodes the input byte-identically, the baseline does not, + * and the input carries the UTF-8 BOM sequence. + */ +const isKnownBomKeepChange = ( + input: Uint8Array, + base: DecodeDiffOutcome, + cur: DecodeDiffOutcome, +): boolean => { + if (!base.ok || !cur.ok) return false; + const inputHex = bytesToHex(input); + return cur.hex === inputHex && base.hex !== inputHex && inputHex.includes("efbbbf"); +}; + +/** + * Decode-behavior change since the baseline: whole-valued f32/f64 heads at or + * beyond the saturating-cast bounds (`fa4f000001`, `fa4f800000`, + * `fb43e0000000000001`, …) are accepted as the reference's + * `validate_canonical_f32/f64` accept them, and decode to the integer node + * `From` builds. The baseline's re-encode check rejected every such + * head as NonCanonicalNumeric. Signature: baseline rejects at the decode stage + * with that code, the working tree either accepts or fails later in the input + * with a different decode-stage code (a corruption mutation appends bytes + * after the head: the baseline stops at the head, the working tree reads past + * it and reports UnusedData), and the input carries a single- or + * double-precision head byte. + */ +const isKnownWholeFloatAcceptChange = ( + input: Uint8Array, + base: DecodeDiffOutcome, + cur: DecodeDiffOutcome, +): boolean => + !base.ok && + base.stage === "decode" && + base.code === "NonCanonicalNumeric" && + (cur.ok || (cur.stage === "decode" && cur.code !== "NonCanonicalNumeric")) && + input.some((b) => b === 0xfa || b === 0xfb); + +/** + * Encode-behavior change since the baseline: `CborDate.fromEpochSeconds(NaN)` + * is the epoch, as the reference's `from_timestamp` saturates it + * (`trunc() as i64` → 0), where the baseline threw InvalidDate. Signature: the + * recipe holds a NaN `date`, the baseline threw that code, the working tree + * encodes. + */ +const isKnownNanDateChange = ( + recipe: Recipe, + base: { ok: true; hex: string } | { ok: false; code: string }, + cur: { ok: true; hex: string } | { ok: false; code: string }, +): boolean => + !base.ok && + base.code === "InvalidDate" && + cur.ok && + JSON.stringify(recipe).includes('{"k":"date","seconds":"NaN"}'); + +const isKnownDecodeChange = ( + input: Uint8Array, + base: DecodeDiffOutcome, + cur: DecodeDiffOutcome, +): boolean => + isKnownLengthCodeChange(base, cur) || + isKnownBomKeepChange(input, base, cur) || + isKnownWholeFloatAcceptChange(input, base, cur); + +/** + * Integrity pin for the baseline. Without this, an accidental re-baseline (or + * a build/copy mishap) silently turns the differential into a + * self-comparison that passes vacuously. Update only when deliberately + * re-baselining, together with tests/baseline/README.md. */ const BASELINE_SHA256 = "ffb0bf6acdafaf01fbb6360497f96cb0f821d4d602f6f343cddd507a302fb72c"; const BASELINE_COMMIT = "187769a273176a2c6089fb412871226b6db1e795"; -const baseline: VectorApi = adapterFor(baselineMod as unknown as Record); -const current: VectorApi = redesignedAdapterFor(src as unknown as Record); +const baseline: VectorApi = baselineAdapterFor(baselineMod as unknown as Record); +const current: VectorApi = currentAdapterFor(src as unknown as Record); /** Stride for the plain decode differential (encode bytes → both decoders). */ const DECODE_STRIDE = 4; @@ -132,7 +186,7 @@ describe("differential corpus: baseline vs working tree", () => { expect(digest, `baseline must be the build from ${BASELINE_COMMIT}`).toBe(BASELINE_SHA256); }); - it("corpus has the exact expected size (~93k plan floor)", () => { + it("corpus has the exact expected size", () => { // Exact equality so a silently shrinking category can't hide inside the // floor. Editing the corpus deliberately => update this constant in the // same reviewed diff. @@ -160,10 +214,10 @@ describe("differential corpus: baseline vs working tree", () => { const cur = encodeOutcome(current, recipe); if (tombstoned) { - // Post-wave: the working tree must throw the directive error (not - // just any error). Most of these encoded in the baseline; a few - // edge recipes (e.g. out-of-range number tags) threw there too - - // either way the tombstone throw must now win. + // The working tree must throw the directive error (not just any + // error). Most of these encoded in the baseline; a few edge recipes + // (e.g. out-of-range number tags) threw there too - either way the + // directive error must win. if (cur.ok) { report(name, `working tree still encodes (${outcomeStr(cur)}) - tombstone missing`); } else if (cur.code !== tombstoneCode) { @@ -178,7 +232,9 @@ describe("differential corpus: baseline vs working tree", () => { (base.ok && cur.ok && base.hex !== cur.hex) || (!base.ok && !cur.ok && base.code !== cur.code) ) { - report(name, `baseline ${outcomeStr(base)} != current ${outcomeStr(cur)}`); + if (!isKnownNanDateChange(recipe, base, cur)) { + report(name, `baseline ${outcomeStr(base)} != current ${outcomeStr(cur)}`); + } index++; continue; } @@ -192,7 +248,7 @@ describe("differential corpus: baseline vs working tree", () => { const curDec = decodeOutcome(current, encoded.slice()); if ( JSON.stringify(baseDec) !== JSON.stringify(curDec) && - !isKnownLengthCodeChange(baseDec, curDec) + !isKnownDecodeChange(encoded, baseDec, curDec) ) { report( name, @@ -201,9 +257,10 @@ describe("differential corpus: baseline vs working tree", () => { } else if (baseDec.ok && baseDec.hex !== base.hex) { report(name, `decode round-trip broke: ${outcomeStr(baseDec)} for ${base.hex}`); } - // NB: both builds REJECTING the encoder's own output identically is - // legal - the frozen bare-Float quirk emits 0xfa floats for whole - // values that the decoder's canonicality re-encode refuses. + // NB: the bare-Float ladder emits 0xfa floats for whole values + // >= 2^32; the baseline decoder refused those as non-canonical, + // the working tree accepts them like the reference - covered by + // isKnownWholeFloatAcceptChange above. if (index % MUTATION_STRIDE === 0) { for (const [mutName, mutated] of mutations(encoded)) { @@ -211,7 +268,7 @@ describe("differential corpus: baseline vs working tree", () => { const curMut = decodeOutcome(current, mutated.slice()); if ( JSON.stringify(baseMut) !== JSON.stringify(curMut) && - !isKnownLengthCodeChange(baseMut, curMut) + !isKnownDecodeChange(mutated, baseMut, curMut) ) { report( `${name}/${mutName}`, @@ -245,7 +302,7 @@ describe("differential corpus: baseline vs working tree", () => { const bytes = hexToBytes(hex); const base = decodeOutcome(baseline, bytes.slice()); const cur = decodeOutcome(current, bytes.slice()); - if (JSON.stringify(base) !== JSON.stringify(cur) && !isKnownLengthCodeChange(base, cur)) { + if (JSON.stringify(base) !== JSON.stringify(cur) && !isKnownDecodeChange(bytes, base, cur)) { failures.push(`${name}: baseline ${outcomeStr(base)} != current ${outcomeStr(cur)}`); } } diff --git a/tests/dist-packaging.test.ts b/tests/dist-packaging.test.ts index 498f642..ccae14e 100644 --- a/tests/dist-packaging.test.ts +++ b/tests/dist-packaging.test.ts @@ -1,14 +1,14 @@ /** - * Dist-level packaging assertions (P3.18). + * Dist-level packaging assertions. * - * Runs against the BUILT `dist/` output (skipped when absent - CI builds - * before testing). The critical invariant: the subpath entries share chunks - * with the root entry, so module-level singletons (the global tags store) - * are one instance across entries. IIFE output was dropped precisely - * because it would fork that singleton. + * Runs against the built `dist/` output (skipped when absent - CI builds + * before testing). The subpath entries share chunks with the root entry, so + * module-level state (the global tags store, the `Cbor` prototype) is one + * instance across entries. */ import { existsSync } from "node:fs"; +import { createRequire } from "node:module"; import { dirname, join } from "node:path"; import { fileURLToPath } from "node:url"; @@ -16,7 +16,7 @@ const here = dirname(fileURLToPath(import.meta.url)); const dist = join(here, "..", "dist"); const built = existsSync(join(dist, "index.mjs")) && existsSync(join(dist, "diagnostic.mjs")); -describe.skipIf(!built)("dist packaging (P3.18)", () => { +describe.skipIf(!built)("dist packaging", () => { it("tags-store singleton is shared across subpath entries", async () => { const root = (await import(join(dist, "index.mjs"))) as { getGlobalTagsStore(): { register(tag: { value: number; name: string }): void }; @@ -33,6 +33,27 @@ describe.skipIf(!built)("dist packaging (P3.18)", () => { expect(rendered).toContain("dist-singleton-probe"); }); + // The global store is keyed on globalThis, so the CommonJS build + // and the ESM build - two module instances - resolve one store, as the + // reference's process-wide GLOBAL_TAGS. + it("global tags store is one instance across the CJS and ESM builds", async () => { + const require = createRequire(import.meta.url); + const cjs = require(join(dist, "index.cjs")) as { + getGlobalTagsStore(): object; + registerStandardTags(): void; + }; + const esm = (await import(join(dist, "index.mjs"))) as { + getGlobalTagsStore(): object; + taggedValue(tag: number, content: unknown): unknown; + }; + const diag = (await import(join(dist, "diagnostic.mjs"))) as { + diagnostic(c: unknown, opts?: { annotate?: boolean }): string; + }; + expect(cjs.getGlobalTagsStore()).toBe(esm.getGlobalTagsStore()); + cjs.registerStandardTags(); + expect(diag.diagnostic(esm.taggedValue(1, 0), { annotate: true })).toMatch(/ date /); + }); + it("debug entry installs hooks onto the shared prototype", async () => { const root = (await import(join(dist, "index.mjs"))) as { cbor(value: unknown): { toString(): string; toJSON?: () => string }; diff --git a/tests/encode.test.ts b/tests/encode.test.ts index 85141ec..9e9afbe 100644 --- a/tests/encode.test.ts +++ b/tests/encode.test.ts @@ -1,13 +1,10 @@ /** - * Encoding tests for dCBOR TypeScript implementation. - * - * This file is a complete 1:1 translation of Rust's tests/encode.rs - * - * All 35 test functions from the Rust version are translated here. + * Encoding tests, ported from Rust's tests/encode.rs, plus TypeScript-specific + * cases for bigint input, text normalization, float heads and equality. */ import type { Cbor, CborInput } from "../src/cbor"; -import { cbor, encodeCbor, taggedValue } from "../src/cbor"; +import { cbor, encodeCbor, taggedValue, cborEquals } from "../src/cbor"; import { diagnostic } from "../src/diag"; import { decodeCbor } from "../src/decode"; import { ByteString } from "../src/byte-string"; @@ -16,7 +13,16 @@ import { CborSet } from "../src/set"; import { CborDate } from "../src/date"; import { Tag } from "../src/tag"; import { CborError } from "../src/error"; -import { extractCbor, asNumber, asInteger, expectInteger } from "../src/conveniences"; +import { + extractCbor, + asNumber, + asInteger, + expectInteger, + expectText, + expectUnsigned, + expectNegative, +} from "../src/conveniences"; +import { hexAnnotated } from "../src/dump"; /** Helper to convert hex string to Uint8Array */ function hexToBytes(hex: string): Uint8Array { @@ -27,18 +33,11 @@ function hexToBytes(hex: string): Uint8Array { return bytes; } -// P3.8: debug-string surface removed - the former `cborDebug` helper (and the -// `expectedDebug` parameter on the test helpers below) asserted the deleted -// debug representation (`map.debug`, Rust-style `{:?}` strings) and was -// dropped with no replacement. - -/** Helper: cborDiagnostic - Get diagnostic description (matches Rust's format!("{}", cbor)) */ +/** Flat diagnostic notation, the counterpart of Rust's `format!("{}", cbor)`. */ function cborDiagnostic(cborValue: Cbor): string { - // Use flat output to match Rust's Display trait (format!("{}", cbor)) return diagnostic(cborValue, { flat: true }); } -/** Helper: cborHex - Get hex encoding */ function cborHex(cborValue: Cbor): string { return cborValue.toHex(); } @@ -120,7 +119,6 @@ function testCborCodable(value: CborInput, expectedDisplay: string, expectedData } describe("encode tests", () => { - // Test 1: encode_unsigned describe("encode_unsigned", () => { test("encode 0", () => { testCborCodable(0, "0", "00"); @@ -158,9 +156,8 @@ describe("encode tests", () => { testCborCodable(4294967296, "4294967296", "1b0000000100000000"); }); - test("encode u64::MAX (JavaScript max safe might differ)", () => { - // JavaScript's Number.MAX_SAFE_INTEGER = 9007199254740991 - // Rust's u64::MAX = 18446744073709551615 (requires BigInt in JS) + test("encode Number.MAX_SAFE_INTEGER (stands in for Rust's u64::MAX)", () => { + // u64::MAX itself needs a bigint; see the bigint boundary tests below. testCborCodable(Number.MAX_SAFE_INTEGER, "9007199254740991", "1b001fffffffffffff"); }); @@ -178,7 +175,6 @@ describe("encode tests", () => { }); }); - // Test 2: encode_signed describe("encode_signed", () => { test("encode -1", () => { testCborCodable(-1, "-1", "20"); @@ -216,13 +212,12 @@ describe("encode tests", () => { testCborCodable(2147483647, "2147483647", "1a7fffffff"); }); - test("encode -9223372036854775808 (i64::MIN - JavaScript safe range)", () => { - // Note: JavaScript's MIN_SAFE_INTEGER is -9007199254740991 - // For full i64 range testing, we'd need BigInt + test("encode Number.MIN_SAFE_INTEGER (stands in for Rust's i64::MIN)", () => { + // i64::MIN itself needs a bigint; see the bigint boundary tests below. testCborCodable(Number.MIN_SAFE_INTEGER, "-9007199254740991", "3b001ffffffffffffe"); }); - test("encode 9223372036854775807 (i64::MAX - JavaScript safe range)", () => { + test("encode Number.MAX_SAFE_INTEGER (stands in for Rust's i64::MAX)", () => { testCborCodable(Number.MAX_SAFE_INTEGER, "9007199254740991", "1b001fffffffffffff"); }); @@ -236,13 +231,11 @@ describe("encode tests", () => { }); }); - // Test 3: encode_bytes_1 test("encode_bytes_1", () => { const bytes = new Uint8Array([0x00, 0x11, 0x22, 0x33]); testCborCodable(new ByteString(bytes), "h'00112233'", "4400112233"); }); - // Test 4: encode_bytes describe("encode_bytes", () => { test("encode 32-byte string", () => { const bytes = hexToBytes("c0a7da14e5847c526244f7e083d26fe33f86d2313ad2b77164233444423a50a7"); @@ -259,7 +252,6 @@ describe("encode tests", () => { }); }); - // Test 5: encode_string describe("encode_string", () => { test('encode "Hello"', () => { testCborCodable("Hello", '"Hello"', "6548656c6c6f"); @@ -276,7 +268,6 @@ describe("encode tests", () => { }); }); - // Test 6: test_normalized_string test("test_normalized_string", () => { const composedEAcute = "\u{00E9}"; // é in NFC const decomposedEAcute = "\u{0065}\u{0301}"; // e + combining acute accent in NFD @@ -303,7 +294,53 @@ describe("encode tests", () => { ); }); - // Test 7: encode_array + // The node keeps the constructed string and the encoder applies NFC, as + // Rust's `CBORCase::Text` and `to_cbor_data` do. Every expected string below + // was executed on the reference. + describe("text nodes keep the constructed string; NFC applies at encode time", () => { + const nfd = "e\u0301"; // "é" decomposed: 65 cc 81 + const nfc = "\u00e9"; // "é" composed: c3 a9 + + test("cbor(string) stores the string verbatim", () => { + const c = cbor(nfd); + expect(expectText(c)).toBe(nfd); + expect(expectText(c)).not.toBe(nfc); + expect(bytesToHex(new TextEncoder().encode(expectText(c)))).toBe("65cc81"); + }); + + test("encodeCbor emits the NFC form", () => { + expect(bytesToHex(encodeCbor(cbor(nfd)))).toBe("62c3a9"); + expect(bytesToHex(encodeCbor(cbor(nfc)))).toBe("62c3a9"); + // U+212B ANGSTROM SIGN is a singleton decomposition to U+00C5. + expect(bytesToHex(encodeCbor(cbor("\u212b")))).toBe("62c385"); + expect(bytesToHex(encodeCbor(cbor("\u00c5")))).toBe("62c385"); + // A bare Text node (no constructor involved) is normalized too. + const bare: CborInput = { isCbor: true, type: 3, value: nfd } as unknown as Cbor; + expect(bytesToHex(encodeCbor(bare))).toBe("62c3a9"); + }); + + test("diagnostic and hexAnnotated show the stored string (Rust dump.rs parity)", () => { + const c = cbor(nfd); + expect(diagnostic(c)).toBe(`"${nfd}"`); + expect(hexAnnotated(c)).toBe(`63 # text(3)\n 65cc81 # "${nfd}"`); + expect(hexAnnotated(cbor(nfc))).toBe(`62 # text(2)\n c3a9 # "${nfc}"`); + }); + + test("map keys compare by encoded bytes, so NFD and NFC are the same key", () => { + const m = new CborMap(); + m.set(nfd, 1); + m.set(nfc, 2); + expect(m.size).toBe(1); + expect(bytesToHex(encodeCbor(m))).toBe("a162c3a902"); + expect(CborSet.from([nfd, nfc]).size).toBe(1); + }); + + test("decoding still rejects non-NFC input", () => { + expect(() => decodeCbor(hexToBytes("6365cc81"))).toThrow(CborError); + expect(decodeCbor(hexToBytes("62c3a9")).value).toBe(nfc); + }); + }); + describe("encode_array", () => { test("encode empty array", () => { testCbor([], "[]", "80"); @@ -318,7 +355,6 @@ describe("encode tests", () => { }); }); - // Test 8: encode_heterogenous_array test("encode_heterogenous_array", () => { const array = [1, "Hello", [1, 2, 3]]; testCbor(array, '[1, "Hello", [1, 2, 3]]', "83016548656c6c6f83010203"); @@ -334,7 +370,6 @@ describe("encode tests", () => { expect(extractedArray[2]).toEqual([1, 2, 3]); }); - // Test 9: encode_map describe("encode_map", () => { test("encode empty map", () => { const m = new CborMap(); @@ -368,7 +403,6 @@ describe("encode tests", () => { }); }); - // Test 10: encode_map_with_map_keys test("encode_map_with_map_keys", () => { const k1 = new CborMap(); k1.set(1, 2); @@ -383,7 +417,6 @@ describe("encode tests", () => { testCbor(m, "{{1: 2}: 5, {3: 4}: 6}", "a2a1010205a1030406"); }); - // Test 11: encode_anders_map test("encode_anders_map", () => { const m = new CborMap(); m.set(1, 45.7); @@ -394,18 +427,15 @@ describe("encode tests", () => { expect(extractCbor(m.getOrThrow(1))).toBe(45.7); }); - // Test 12: encode_map_misordered test("encode_map_misordered", () => { expect(() => decodeCbor(hexToBytes("a2026141016142"))).toThrow(/canonical order/i); }); - // Test 13: encode_tagged test("encode_tagged", () => { const tagged = taggedValue(1, "Hello"); testCbor(tagged, '1("Hello")', "c16548656c6c6f"); }); - // Test 14: encode_value describe("encode_value", () => { test("encode false", () => { testCbor(false, "false", "f4"); @@ -416,10 +446,7 @@ describe("encode tests", () => { }); }); - // Test 15: encode_envelope test("encode_envelope", () => { - // P3.5: {tag, value} literal inputs no longer encode as tagged values; - // build tagged values explicitly with taggedValue() (identical bytes). const alice = taggedValue(200, taggedValue(201, "Alice")); const knows = taggedValue(200, taggedValue(201, "knows")); const bob = taggedValue(200, taggedValue(201, "Bob")); @@ -439,8 +466,8 @@ describe("encode tests", () => { expect(encodeCbor(envelope)).toEqual(encodeCbor(decodedCbor)); }); - // P3.5: the {tag, value} key-sniffing input mapping is removed - a plain - // two-key {tag, value} literal now throws CborError with code "Custom". + // A plain two-key {tag, value} literal is rejected rather than taken for a + // tagged value; tagged values are built with taggedValue(). test("plain {tag, value} literal throws CborError code Custom", () => { let error: unknown; try { @@ -452,8 +479,8 @@ describe("encode tests", () => { expect(CborError.isCborError(error) && error.code).toBe("Custom"); }); - // Test 16: encode_float - full 1:1 port of Rust encode.rs `encode_float` - // (all ~30 vectors, including the numeric-reduction boundary cliffs). + // Port of Rust's `encode_float`, including the numeric-reduction boundary + // cliffs. describe("encode_float", () => { test("shortest accurate representation", () => { testCbor(1.5, "1.5", "f93e00"); @@ -526,10 +553,9 @@ describe("encode tests", () => { testCbor(1.7976931348623157e308, "1.7976931348623157e308", "fb7fefffffffffffff"); }); - test("large whole-valued JS numbers reduce to integers like the bigint path (C1)", () => { - // Regression for C1: a whole `number` beyond the JS safe-integer range - // must encode as an integer (matching Rust `From` and the bigint - // path), not as a float. + test("large whole-valued JS numbers reduce to integers like the bigint path", () => { + // A whole `number` beyond the safe-integer range encodes as an integer, + // as Rust's `From` and the bigint path do, not as a float. expect(cbor(2 ** 53).toHex()).toBe("1b0020000000000000"); expect(cbor(2 ** 53).toHex()).toBe(cbor(9007199254740992n).toHex()); expect(cbor(2 ** 63).toHex()).toBe("1b8000000000000000"); @@ -537,7 +563,164 @@ describe("encode tests", () => { }); }); - // Test 17: int_coerced_to_float + // Float heads are judged by the reference's + // `validate_canonical_f16/f32/f64` predicates (saturating `as i32`/`as i64` + // casts) and decode to the node `From`/`From` builds. Every row + // was executed on dcbor 0.25.2. + describe("decoding whole-valued f32/f64 heads (Rust validate_canonical_* parity)", () => { + const dec = (hex: string) => decodeCbor(hexToBytes(hex)); + + test("f32 wholes at or beyond 2^31 are accepted and reduce to integers", () => { + const c = dec("fa4f000001"); // 2^31 + 256 + expect(c.type).toBe(0); + expect(expectUnsigned(c)).toBe(2147483904); + expect(c.toHex()).toBe("1a80000100"); + expect(dec("fa4f7fffff").toHex()).toBe("1affffff00"); // 2^32 - 256 + expect(dec("facf000001").toHex()).toBe("3a80000100"); // -(2^31 + 256) -> -2147483905 + expect(cborDiagnostic(dec("facf000001"))).toBe("-2147483905"); + }); + + test("negative f32 magnitudes are computed in f32 arithmetic (-1f32 - n)", () => { + const c = dec("fadf000000"); // -2^63 + expect(c.toHex()).toBe("3b8000000000000000"); + expect(cborDiagnostic(c)).toBe("-9223372036854775809"); + expect(expectNegative(c)).toBe(-9223372036854775809n); + const d = dec("fadb000000"); // -2^55 + expect(d.toHex()).toBe("3b0080000000000000"); + expect(cborDiagnostic(d)).toBe("-36028797018963969"); + }); + + test("f32 wholes with no fitting integer stay floats and round-trip", () => { + for (const hex of ["fa4f800000", "fa4fc00000", "fa5a000000", "fa5f000000", "fadf800000"]) { + const c = dec(hex); + expect(c.type).toBe(7); + expect(c.toHex()).toBe(hex); + } + expect(extractCbor(dec("fa4f800000"))).toBe(4294967296); + }); + + test("f64 wholes beyond the i64 saturation bound are accepted", () => { + expect(dec("fb43e0000000000001").toHex()).toBe("1b8000000000000800"); // 2^63 + 2048 + expect(dec("fbc3e0000000000001").toHex()).toBe("3b80000000000007ff"); // -(2^63 + 2048) + }); + + test("the exact saturation bounds are still non-canonical", () => { + for (const hex of ["fa4f000000", "facf000000", "fb43e0000000000000", "fbc3e0000000000000"]) { + let error: unknown; + try { + dec(hex); + } catch (e) { + error = e; + } + expect(CborError.isCborError(error) && error.code, hex).toBe("NonCanonicalNumeric"); + } + }); + + test("f16 heads keep the same accept set", () => { + expect(dec("f97e00").toHex()).toBe("f97e00"); // canonical NaN + expect(dec("f93e00").toHex()).toBe("f93e00"); // 1.5 + for (const hex of ["f97e01", "f94200", "f98000", "f90000"]) { + expect(() => dec(hex), hex).toThrow(CborError); + } + }); + }); + + // `cborEquals` is the reference's `PartialEq for CBOR`, not a + // byte comparison. Every row was executed on dcbor 0.25.2 (`==`). + describe("cborEquals is structural (Rust PartialEq parity)", () => { + const float = (v: number): Cbor => + cbor({ isCbor: true, type: 7, value: { type: "Float", value: v } } as unknown as Cbor); + + test("a whole-valued float node is not the integer it encodes as", () => { + expect(encodeCbor(float(2))).toEqual(encodeCbor(cbor(2))); + expect(cborEquals(float(2), cbor(2))).toBe(false); + expect(cborEquals(float(-0), cbor(0))).toBe(false); + expect(cborEquals(float(-1), cbor(-1))).toBe(false); + expect(cborEquals(cbor([float(1)]), cbor([1]))).toBe(false); + }); + + test("floats compare by value, with NaN equal to NaN", () => { + expect(cborEquals(float(-0), float(0))).toBe(true); + expect(cborEquals(float(NaN), float(NaN))).toBe(true); + expect(cborEquals(cbor(1.5), cbor(1.5))).toBe(true); + expect(cborEquals(cbor(1.5), cbor(2.5))).toBe(false); + }); + + test("integers compare by value across number and bigint", () => { + expect(cborEquals(cbor(5), cbor(5n))).toBe(true); + expect(cborEquals(cbor(-5), cbor(-5n))).toBe(true); + expect(cborEquals(cbor(2n ** 63n), cbor(2 ** 63))).toBe(true); + expect(cborEquals(cbor(5), cbor(6))).toBe(false); + expect(cborEquals(cbor(5), cbor(-5))).toBe(false); + }); + + test("text compares the stored string, so NFD is not NFC", () => { + expect(cborEquals(cbor("e\u0301"), cbor("\u00e9"))).toBe(false); + expect(cborEquals(cbor("\u00e9"), cbor("\u00e9"))).toBe(true); + expect(cborEquals(cbor("ab"), cbor(new TextEncoder().encode("ab")))).toBe(false); + }); + + test("byte strings compare bytewise", () => { + expect(cborEquals(cbor(new Uint8Array([1, 2])), cbor(new Uint8Array([1, 2])))).toBe(true); + expect(cborEquals(cbor(new Uint8Array([1, 2])), cbor(new Uint8Array([1, 2, 3])))).toBe(false); + }); + + test("tags compare by value only; the carried name is ignored", () => { + expect(cborEquals(taggedValue(Tag.from(1, "date"), 0), taggedValue(1, 0))).toBe(true); + expect(cborEquals(taggedValue(1, 0), taggedValue(2, 0))).toBe(false); + expect(cborEquals(taggedValue(1, 0), taggedValue(1, 1))).toBe(false); + }); + + test("maps compare entry by entry in canonical order, keys and values structurally", () => { + const ab = new CborMap(); + ab.set("a", 1); + ab.set("b", 2); + const ba = new CborMap(); + ba.set("b", 2); + ba.set("a", 1); + expect(cborEquals(cbor(ab), cbor(ba))).toBe(true); + const a = new CborMap(); + a.set("a", 1); + expect(cborEquals(cbor(ab), cbor(a))).toBe(false); + const floatKey = new CborMap(); + floatKey.set(float(1), "x"); + const intKey = new CborMap(); + intKey.set(1, "x"); + expect(encodeCbor(floatKey)).toEqual(encodeCbor(intKey)); + expect(cborEquals(cbor(floatKey), cbor(intKey))).toBe(false); + const nfdKey = new CborMap(); + nfdKey.set("e\u0301", 1); + const nfcKey = new CborMap(); + nfcKey.set("\u00e9", 1); + expect(cborEquals(cbor(nfdKey), cbor(nfcKey))).toBe(false); + }); + + test("nested structures, decoded vs constructed", () => { + const ab = new CborMap(); + ab.set("a", 1); + ab.set("b", 2); + const nested = taggedValue(5, [ab, [1, 1.5]]); + expect(cborEquals(nested, decodeCbor(encodeCbor(nested)))).toBe(true); + expect(cborEquals(nested, taggedValue(5, [ab, [1, 2.5]]))).toBe(false); + expect(cborEquals(nested, taggedValue(6, [ab, [1, 1.5]]))).toBe(false); + }); + }); + + // Nesting is bounded only by the host stack on both sides; 1,000 levels + // must work everywhere. + test("a 1,000-deep array decodes, re-encodes identically and renders", () => { + const depth = 1000; + const bytes = new Uint8Array(depth + 1); + bytes.fill(0x81, 0, depth); + bytes[depth] = 0x00; + const decoded = decodeCbor(bytes); + expect(encodeCbor(decoded)).toEqual(bytes); + const flat = diagnostic(decoded, { flat: true }); + expect(flat).toBe(`${"[".repeat(depth)}0${"]".repeat(depth)}`); + // Every level but the innermost `[0]` opens and closes on its own line. + expect(diagnostic(decoded).split("\n").length).toBe(2 * depth - 1); + }); + test("int_coerced_to_float", () => { const n = 42; const c = cbor(n); @@ -549,7 +732,6 @@ describe("encode tests", () => { expect(i).toBe(n); }); - // Test 18: fail_float_coerced_to_int test("fail_float_coerced_to_int", () => { // Floating point values cannot be coerced to integer types (mirrors Rust // `i32::try_from(c)` returning Err for a fractional float). @@ -562,22 +744,18 @@ describe("encode tests", () => { expect(() => expectInteger(c)).toThrow(); }); - // Test 19: non_canonical_float_1 test("non_canonical_float_1", () => { expect(() => decodeCbor(hexToBytes("FB3FF8000000000000"))).toThrow(/canonical/i); }); - // Test 20: non_canonical_float_2 test("non_canonical_float_2", () => { expect(() => decodeCbor(hexToBytes("F94A00"))).toThrow(/canonical/i); }); - // Test 21: unused_data test("unused_data", () => { expect(() => decodeCbor(hexToBytes("0001"))).toThrow(/extra bytes/i); }); - // Test 22: tag test("tag", () => { const tag = Tag.from(1, "A"); expect(tag.name).toBe("A"); @@ -588,12 +766,10 @@ describe("encode tests", () => { expect(tag2.value).toBe(2); }); - // Test 23: encode_date test("encode_date", () => { testCborCodable(CborDate.fromEpochSeconds(1675854714.0), "1(1675854714)", "c11a63e3837a"); }); - // Test 24: convert_values describe("convert_values", () => { function testConvert(value: CborInput) { const cborValue = cbor(value); @@ -621,7 +797,7 @@ describe("encode tests", () => { test("convert ByteString", () => testConvert(new ByteString(hexToBytes("001122334455")))); }); - // Test 25: convert_hash_map (TypeScript uses Map) + // Rust's HashMap and BTreeMap tests both use a JS Map. test("convert_hash_map", () => { const h = new Map(); h.set(1, "A"); @@ -638,7 +814,6 @@ describe("encode tests", () => { expect(h2.get(50)).toBe("B"); }); - // Test 26: convert_btree_map (same as hash_map in TypeScript) test("convert_btree_map", () => { const h = new Map(); h.set(1, "A"); @@ -655,7 +830,7 @@ describe("encode tests", () => { expect(h2.get(50)).toBe("B"); }); - // Test 27: convert_vector + // Rust's Vec and VecDeque tests both use a JS array. test("convert_vector", () => { const v = [1, 50, 25]; const c = cbor(v); @@ -665,7 +840,6 @@ describe("encode tests", () => { expect(v2).toEqual(v); }); - // Test 28: convert_vecdeque (TypeScript arrays work the same) test("convert_vecdeque", () => { const v = [1, 50, 25]; const c = cbor(v); @@ -675,7 +849,6 @@ describe("encode tests", () => { expect(v2).toEqual(v); }); - // Test 29: convert_hashset (TypeScript Set) test("convert_hashset", () => { const v = new Set([1, 50, 25]); const c = cbor(v); @@ -686,14 +859,12 @@ describe("encode tests", () => { expect(v2.has(25)).toBe(true); }); - // Test 30: usage_test_1 test("usage_test_1", () => { const array = [1000, 2000, 3000]; const cborValue = cbor(array); expect(cborHex(cborValue)).toBe("831903e81907d0190bb8"); }); - // Test 31: usage_test_2 test("usage_test_2", () => { const data = hexToBytes("831903e81907d0190bb8"); const cborValue = decodeCbor(data); @@ -703,7 +874,6 @@ describe("encode tests", () => { expect(array).toEqual([1000, 2000, 3000]); }); - // Test 32: encode_nan test("encode_nan", () => { const canonicalNanData = hexToBytes("f97e00"); @@ -714,7 +884,6 @@ describe("encode tests", () => { expect(encodeCbor(cbor(NaN))).toEqual(canonicalNanData); }); - // Test 33: decode_nan test("decode_nan", () => { // Canonical NaN decodes const canonicalNanData = hexToBytes("f97e00"); @@ -728,7 +897,7 @@ describe("encode tests", () => { expect(() => decodeCbor(hexToBytes("fb7ff9100000000001"))).toThrow(); }); - // Test 34: encode_infinit (typo preserved from Rust) + // The name keeps the Rust test's spelling. test("encode_infinit", () => { const canonicalInfinityData = hexToBytes("f97c00"); const canonicalNegInfinityData = hexToBytes("f9fc00"); @@ -737,7 +906,6 @@ describe("encode tests", () => { expect(encodeCbor(cbor(-Infinity))).toEqual(canonicalNegInfinityData); }); - // Test 35: decode_infinity test("decode_infinity", () => { const canonicalInfinityData = hexToBytes("f97c00"); const canonicalNegInfinityData = hexToBytes("f9fc00"); @@ -758,8 +926,8 @@ describe("encode tests", () => { expect(() => decodeCbor(hexToBytes("fbfff0000000000000"))).toThrow(); }); - // Set parity with Rust `dcbor` (C5): a Set is a PLAIN UNTAGGED ARRAY in - // strict ascending CBOR-byte order with no duplicates - NOT tag-258. + // As in Rust `dcbor`, a Set is an untagged array in strictly ascending + // CBOR-byte order with no duplicates (no tag 258). describe("set (untagged array, Rust parity)", () => { test("encodes as an untagged array through every path", () => { const set = CborSet.from([3, 1, 2]); @@ -797,9 +965,9 @@ describe("encode tests", () => { expect(() => CborSet.fromCbor(c)).toThrow(); }); - test("does NOT emit a tag-258 wrapper", () => { + test("does not emit a tag-258 wrapper", () => { const set = CborSet.from([1, 2, 3]); - // The old (incorrect) tag-258 encoding was d9010283010203. + // d90102 is the tag-258 head. expect(bytesToHex(set.toBytes())).not.toContain("d90102"); }); }); diff --git a/tests/error.test.ts b/tests/error.test.ts index 21481d1..a60e476 100644 --- a/tests/error.test.ts +++ b/tests/error.test.ts @@ -237,8 +237,8 @@ describe("CborError", () => { } }); - // Regression for C3: invalid UTF-8 must be rejected, not decoded to U+FFFD. - test("throws InvalidUtf8 for invalid UTF-8 text strings (C3)", () => { + // Invalid UTF-8 is rejected, not decoded to U+FFFD. + test("throws InvalidUtf8 for invalid UTF-8 text strings", () => { const badUtf8 = new Uint8Array([0x62, 0xc3, 0x28]); expect(() => decodeCbor(badUtf8)).toThrow(CborError); try { @@ -249,6 +249,58 @@ describe("CborError", () => { expect(() => decodeCbor(new Uint8Array([0x61, 0xff]))).toThrow(CborError); }); + // The message is the reference's `Utf8Error` Display, computed + // from the bytes, not the host decoder's text. Every row was executed on + // dcbor 0.25.2 (`CBOR::try_from_data(..).unwrap_err().to_string()`). + test("InvalidUtf8 messages mirror core::str::Utf8Error", () => { + const invalid = (n: number, i: number) => + `invalid utf-8 sequence of ${n} bytes from index ${i}`; + const incomplete = (i: number) => `incomplete utf-8 byte sequence from index ${i}`; + const rows: [string, string][] = [ + ["62c328", invalid(1, 0)], + ["62c080", invalid(1, 0)], + ["63eda080", invalid(1, 0)], + ["61c3", incomplete(0)], + ["64f4908080", invalid(1, 0)], + ["61ff", invalid(1, 0)], + ["8162c328", invalid(1, 0)], + ["c162c328", invalid(1, 0)], + ["63e28228", invalid(2, 0)], + ["64f09f9841", invalid(3, 0)], + ["63e0a041", invalid(2, 0)], + ["62f0a0", incomplete(0)], + ["6441c3a9ff", invalid(1, 3)], + ["64616263ff", invalid(1, 3)], + ["62eda0", invalid(1, 0)], // a present bad byte beats end of input + ["62c0af", invalid(1, 0)], + ["61e0", incomplete(0)], + ["62e0a0", incomplete(0)], + ["63f0908f", incomplete(0)], + ["62f480", incomplete(0)], + ["61f5", invalid(1, 0)], + ["61c2", incomplete(0)], + ["6180", invalid(1, 0)], // a lone continuation byte + ["6461626380", invalid(1, 3)], + ]; + for (const [hex, suffix] of rows) { + let error: unknown; + try { + decodeCbor(hexToBytes(hex)); + } catch (e) { + error = e; + } + expect(CborError.isCborError(error) && error.code, hex).toBe("InvalidUtf8"); + expect(CborError.isCborError(error) && error.message, hex).toBe( + `invalid UTF\u20118 string: ${suffix}`, + ); + expect(CborError.isCborError(error) && error.details.cause, hex).toBe(suffix); + } + expect(decodeCbor(hexToBytes("65e282acc380")).value).toBe("€À"); + // Noncharacters and the last code point are valid UTF-8 (as in Rust). + expect(decodeCbor(hexToBytes("63efbfbe")).toHex()).toBe("63efbfbe"); + expect(decodeCbor(hexToBytes("64f48fbfbf")).toHex()).toBe("64f48fbfbf"); + }); + test("accepts valid UTF-8 (incl. multibyte) text strings", () => { const ok = new Uint8Array([0x62, 0xc3, 0xa9]); // "é" NFC const c = decodeCbor(ok); @@ -256,8 +308,7 @@ describe("CborError", () => { expect(c.value).toBe("é"); }); - // Regression for C4: interleaved-misordered map keys must be rejected. - test("throws MisorderedMapKey for interleaved-misordered map keys (C4)", () => { + test("throws MisorderedMapKey for interleaved-misordered map keys", () => { const interleaved = hexToBytes("a3010103030202"); expect(() => decodeCbor(interleaved)).toThrow(CborError); try { @@ -267,14 +318,14 @@ describe("CborError", () => { } }); - test("accepts a canonically-ordered 3-key map (C4 control)", () => { + test("accepts a canonically-ordered 3-key map", () => { const ordered = hexToBytes("a3010102020303"); expect(decodeCbor(ordered).type).toBe(5); }); - // Regression for M3: a truncated inner item surfaces as Underrun even when - // the input is a sub-array of a larger ArrayBuffer. - test("does not read past the logical end of a sub-array input (M3)", () => { + // A truncated inner item is Underrun even when the input is a sub-array of + // a larger ArrayBuffer. + test("does not read past the logical end of a sub-array input", () => { const backing = hexToBytes("8201ffff"); const logical = backing.subarray(0, 2); // only `82 01` expect(() => decodeCbor(logical)).toThrow(CborError); @@ -285,7 +336,7 @@ describe("CborError", () => { } }); - test("decodes a valid item that is a sub-array of a larger buffer (M3 control)", () => { + test("decodes a valid item that is a sub-array of a larger buffer", () => { const backing = hexToBytes("820102ffff"); const logical = backing.subarray(0, 3); // `82 01 02` expect(decodeCbor(logical).type).toBe(4); diff --git a/tests/exact.test.ts b/tests/exact.test.ts index 53c654b..8000860 100644 --- a/tests/exact.test.ts +++ b/tests/exact.test.ts @@ -2,8 +2,7 @@ * Exact-conversion tests - 1:1 port of Rust's `src/exact.rs` `mod tests`. * * The `Exact*` helpers underpin dCBOR's numeric reduction (float→int and - * width-narrowing) and were previously untested on the TypeScript side. Each - * case asserts the same boundary behavior as the reference: exact integers + * width-narrowing). Each case asserts the same boundary behavior as the reference: exact integers * convert, fractional/NaN/Infinity/out-of-range inputs reject (undefined), and * float round-trips use the same saturating-cast semantics as Rust's `as`. * @@ -41,6 +40,11 @@ describe("Exact conversions (port of Rust exact.rs tests)", () => { expect(ExactI16.exactFromF16(NaN)).toBeUndefined(); expect(ExactI16.exactFromF16(Infinity)).toBeUndefined(); expect(ExactI16.exactFromF16(-Infinity)).toBeUndefined(); + // The reference excludes i16::MIN from the f16 form only. + expect(ExactI16.exactFromF16(-32768)).toBeUndefined(); + expect(ExactI16.exactFromF16(32767)).toBe(32767); + expect(ExactI16.exactFromF32(-32768)).toBe(-32768); + expect(ExactI16.exactFromF64(-32768)).toBe(-32768); expect(ExactI16.exactFromF32(21.0)).toBe(21); expect(ExactI16.exactFromF32(21.5)).toBeUndefined(); diff --git a/tests/format.test.ts b/tests/format.test.ts index 04848b9..f158a68 100644 --- a/tests/format.test.ts +++ b/tests/format.test.ts @@ -1,21 +1,24 @@ /** - * Format tests - 1:1 translation from Rust's tests/format.rs + * Format tests, ported from Rust's tests/format.rs: diagnostic notation + * (pretty, annotated, flat), summary, and hex (plain and annotated). * - * Tests various formatting outputs including: - * - Diagnostic notation (pretty, annotated, and flat) - * - Summary format - * - Hex encoding (plain and annotated) - * - * P3.8: the Display (`description`) and debug-string (`debug_description`) - * surfaces were removed with no replacement; those assertions are gone. - * Where the old expected description equalled the flat diagnostic, that - * exact string is still asserted via the flat-diagnostic parameter. + * Rust's `description` and `debug_description` outputs have no TypeScript + * counterpart and are not asserted. */ import type { CborInput } from "../src"; -import { cbor, CborMap, registerStandardTags, CborDate, decodeCbor, taggedValue } from "../src"; +import { + cbor, + CborMap, + registerStandardTags, + CborDate, + decodeCbor, + taggedValue, + simpleName, +} from "../src"; import { diagnostic } from "../src/diag"; import { hexAnnotated } from "../src/dump"; +import { floatDisplayString } from "../src/float"; /** Helper to convert a hex string to a Uint8Array. */ function hexToBytes(hexStr: string): Uint8Array { @@ -27,7 +30,7 @@ function hexToBytes(hexStr: string): Uint8Array { } // Compare one formatted output against its expectation; an empty expectation -// just logs the actual output (matches the original helper's behavior). +// just logs the actual output. function check(testName: string, label: string, actual: string, expected: string) { if (expected === "") { console.log(`${label}:`); @@ -42,8 +45,7 @@ function check(testName: string, label: string, actual: string, expected: string expect(actual).toBe(expected); } -// Main test runner function - matches Rust's run() function -// P3.8: description (Display) and debug-string parameters removed. +// Counterpart of Rust's `run()`, without the description and debug outputs. function run( testName: string, value: CborInput, @@ -236,7 +238,6 @@ describe("format tests", () => { }); test("format_tagged", () => { - // Create tagged CBOR: tag 100 with value "Hello" const tagged = taggedValue(100, "Hello"); run( "format_tagged", @@ -306,8 +307,6 @@ describe("format tests", () => { "d83183015829536f6d65206d7973746572696573206172656e2774206d65616e7420746f20626520736f6c7665642e82d902c3820158402b9238e19eafbc154b49ec89edd4e0fb1368e97332c6913b4beb637d1875824f3e43bd7fb0c41fb574f08ce00247413d3ce2d9466e0ccfa4a89b92504982710ad902c3820158400f9c7af36804ffe5313c00115e5a31aa56814abaa77ff301da53d48613496e9c51a98b36d55f6fb5634fdb0123910cfa4904f1c60523df41013dc3749b377900"; const cborValue = decodeCbor(hexToBytes(encodedCborHex)); - // P3.8: description (Display) and debug-string surfaces removed; the old - // expected description equalled the flat diagnostic asserted below. const diagnosticStr = `49( [ 1, @@ -412,19 +411,15 @@ describe("format tests", () => { 78 7b # text(123) 4c6f72656d20697073756d20646f6c6f722073697420616d65742c20636f6e73656374657475722061646970697363696e6720656c69742c2073656420646f20656975736d6f642074656d706f7220696e6369646964756e74207574206c61626f726520657420646f6c6f7265206d61676e6120616c697175612e # "Lorem ipsum dolor sit amet, consectetur adipiscing elit, sed do eiusmod tempor incididunt ut labore et dolore magna aliqua."`; - // Assert the LIBRARY outputs directly. The value is decoded (its tags carry - // no name), so - exactly like Rust - the plain diagnostic path renders tag - // numbers (`1(...)`), while only the annotated path resolves `/ date /` and - // the summary renders `2021-02-24`. + // The decoded tags carry no name, so, as in Rust, the plain diagnostic + // prints tag numbers (`1(...)`), the annotated form resolves `/ date /` and + // the summary prints `2021-02-24`. expect(diagnostic(cborValue)).toBe(diagnosticStr); expect(diagnostic(cborValue, { annotate: true })).toBe(diagnosticAnnotated); expect(diagnostic(cborValue, { flat: true })).toBe(diagnosticFlat); expect(diagnostic(cborValue, { summarize: true })).toBe(summaryStr); expect(cborValue.toHex()).toBe(hex); expect(hexAnnotated(cborValue)).toBe(hexAnnotatedStr); - // P3.8: debug-string surface removed (old expected: - // 'tagged(300, map({...}))'); Display description equalled the flat - // diagnostic asserted above. }); test("format_key_order", () => { @@ -438,8 +433,6 @@ describe("format tests", () => { m.set("aa", 5); m.set([100], 6); - // P3.8: description (Display) and debug-string surfaces removed; the old - // expected description equalled the flat diagnostic asserted below. const diagnosticStr = `{ 10: 1, @@ -495,7 +488,99 @@ describe("format tests", () => { }); }); -describe("diagnostic line breaking measures strings in UTF-8 bytes (review N1)", () => { +describe("float diagnostic matches Rust {:?} on exact decimal ties", () => { + // JS `String()` rounds an exact decimal tie to the even digit; Rust's + // flt2dec rounds the magnitude up. Every expected string below was + // executed on the reference (`format!("{:?}", f64)`). + const dec = (hex: string) => decodeCbor(hexToBytes(hex)); + + it("rounds ties up in exponential notation", () => { + expect(diagnostic(dec("f9000a"))).toBe("5.960464477539063e-7"); // 10 * 2^-24 + expect(diagnostic(dec("f90032"))).toBe("2.9802322387695313e-6"); // 50 * 2^-24 + expect(diagnostic(dec("fa33000000"))).toBe("2.9802322387695313e-8"); // 2^-25 + expect(diagnostic(cbor(-(10 * 2 ** -24)))).toBe("-5.960464477539063e-7"); + }); + + it("rounds ties up in decimal notation", () => { + expect(diagnostic(dec("fb4090000010000000"))).toBe("1024.0000610351563"); // 1024 + 2^-14 + expect(diagnostic(dec("fb4210000000000800"))).toBe("17179869184.007813"); // 2^34 + 2^-7 + expect(diagnostic(cbor(-(1024 + 2 ** -14)))).toBe("-1024.0000610351563"); + expect(hexAnnotated(dec("fb4090000010000000"))).toBe( + "fb4090000010000000 # 1024.0000610351563", + ); + }); + + it("leaves non-tie values and the notation thresholds unchanged", () => { + expect(diagnostic(cbor(0.5 + 2 ** -53))).toBe("0.5000000000000001"); + expect(diagnostic(cbor(1 + 3 * 2 ** -52))).toBe("1.0000000000000007"); + expect(diagnostic(cbor(123456789.125))).toBe("123456789.125"); + expect(diagnostic(cbor(4.35))).toBe("4.35"); + expect(diagnostic(cbor(5e-324))).toBe("5e-324"); + expect(diagnostic(cbor(0.0001))).toBe("0.0001"); + expect(diagnostic(cbor(0.00001))).toBe("1e-5"); + expect(diagnostic(cbor(1e21))).toBe("1e21"); + expect(diagnostic(cbor(1.5e20))).toBe("1.5e20"); + expect(diagnostic(cbor(-0.0))).toBe("0"); // -0.0 integer-reduces + expect(diagnostic(cbor(1.7976931348623157e308))).toBe("1.7976931348623157e308"); + // Whole values integer-reduce through cbor(); the float renderer itself + // follows Rust's `{:?}` thresholds: decimal below 1e16, exponential from it. + expect(floatDisplayString(1.5e15)).toBe("1500000000000000.0"); + expect(floatDisplayString(2 ** 53)).toBe("9007199254740992.0"); + expect(floatDisplayString(1e16)).toBe("1e16"); + expect(floatDisplayString(-1e16)).toBe("-1e16"); + }); + + it("simpleName renders floats like Rust's Simple Debug", () => { + expect(simpleName({ type: "Float", value: Infinity })).toBe("inf"); + expect(simpleName({ type: "Float", value: -Infinity })).toBe("-inf"); + expect(simpleName({ type: "Float", value: NaN })).toBe("NaN"); + expect(simpleName({ type: "Float", value: 1.5 })).toBe("1.5"); + expect(simpleName({ type: "Float", value: 42 })).toBe("42.0"); + expect(simpleName({ type: "Float", value: 10 * 2 ** -24 })).toBe("5.960464477539063e-7"); + expect(simpleName({ type: "True" })).toBe("true"); + }); +}); + +describe("byte-string notes treat every non-ASCII code point as printable", () => { + // Rust's `is_printable(c: char)` sees whole code points, so astral + // characters are printable too. Executed on dcbor 0.25.2. + it("keeps astral characters in the note", () => { + expect(hexAnnotated(cbor(hexToBytes("f09f9880")))).toBe( + `44 # bytes(4)\n f09f9880 # "😀"`, + ); + expect(hexAnnotated(cbor(hexToBytes("00f09f9880")))).toBe( + `45 # bytes(5)\n 00f09f9880 # ".😀"`, + ); + expect(hexAnnotated(cbor(hexToBytes("f0908080e29c93")))).toBe( + `47 # bytes(7)\n f0908080e29c93 # "𐀀✓"`, + ); + }); + it("still dots ASCII controls and omits the note when nothing is printable", () => { + expect(hexAnnotated(cbor(hexToBytes("0a41")))).toBe( + `42 # bytes(2)\n 0a41 # ".A"`, + ); + expect(hexAnnotated(cbor(hexToBytes("00")))).toBe(`41 # bytes(1)\n 00`); + }); +}); + +describe("text decoding keeps a leading U+FEFF", () => { + // Rust's `String::from_utf8` preserves every code point, including a leading + // byte-order mark that the WHATWG TextDecoder strips by default. + it("decodes a BOM-prefixed text string to the same bytes", () => { + const decoded = decodeCbor(hexToBytes("64efbbbf61")); + expect(decoded.type).toBe(3); + expect(decoded.value).toBe("\uFEFFa"); + expect(decoded.toHex()).toBe("64efbbbf61"); + expect(decodeCbor(hexToBytes("63efbbbf")).toHex()).toBe("63efbbbf"); + }); + it("annotates a BOM-prefixed byte string with the BOM in the note", () => { + expect(hexAnnotated(cbor(hexToBytes("efbbbf41")))).toBe( + `44 # bytes(4)\n efbbbf41 # "\uFEFFA"`, + ); + }); +}); + +describe("diagnostic line breaking measures strings in UTF-8 bytes", () => { // `diag.rs`: a group breaks when it contains a group, or its strings total // more than 20 *bytes*, or its greatest child does. Executed on the // reference for every row below. diff --git a/tests/golden-vectors.test.ts b/tests/golden-vectors.test.ts index 3c8570b..287fa44 100644 --- a/tests/golden-vectors.test.ts +++ b/tests/golden-vectors.test.ts @@ -1,20 +1,21 @@ /** - * Golden wire-vector suite (API_REDESIGN_PLAN P1.1a) - the committed, - * hand-pinned freeze of the deterministic wire format. + * Golden wire-vector suite: the committed, hand-reviewed record of the + * deterministic wire format. * * Verifies the working tree against the committed fixtures: - * - tests/vectors/encode-vectors.json: construction recipe → expected bytes - * (or expected CborError code for inputs that throw) - * - tests/vectors/decode-vectors.json: bytes → accept (byte-identical - * re-encode) or reject with a specific CborError.code - covering every - * reachable decoder throw site, including the checkCanonicalEncoding - * re-encode rejections + * - encode-vectors.json: construction recipe → expected bytes (or expected + * CborError code for inputs that throw) + * - decode-vectors.json: bytes → accept (re-encoded bytes) or reject with a + * specific CborError code and message - covering every reachable decoder + * throw site, including the checkCanonicalEncoding re-encode rejections + * - format-vectors.json, date-vectors.json, uint-vectors.json: diagnostic + * renderings, CborDate decode/display, and fixed-width unsigned extraction * * Unlike tests/golden.test.ts (vitest snapshots, auto-updatable with -u), * these fixtures only change through a deliberate run of * `bun run vectors:generate` - the diff is the reviewable record of any - * wire-format change. During the API redesign the wire is FROZEN: any diff - * here outside the two planned tombstone flips (P3.5 / P3.7) is a bug. + * wire-format change. The Rust harness (tests/rust-validation) checks the + * same fixtures against the reference. */ import { createHash } from "node:crypto"; @@ -24,11 +25,27 @@ import { fileURLToPath } from "node:url"; import * as src from "../src"; import { - redesignedAdapterFor, + cbor, + type CborInput, + CborDate, + decodeCbor, + encodeCbor, + expectUnsigned, + registerStandardTags, + TagsStore, +} from "../src"; +import { diagnostic } from "../src/diag"; +import { hexAnnotated } from "../src/dump"; +import { + currentAdapterFor, + bytesToHex, + decodeErrorMessage, decodeOutcome, encodeOutcome, hexToBytes, + materialize, type Recipe, + type RemovedInputShape, } from "./vectors/recipes"; const here = dirname(fileURLToPath(import.meta.url)); @@ -40,23 +57,43 @@ interface EncodeFixture { | { ok: true; hex: string; decodeRejects?: string } | { ok: true; byteLength: number; sha256: string; hexPrefix: string; decodeRejects?: string } | { ok: false; code: string }; - tombstone?: "P3.5" | "P3.7"; + tombstone?: RemovedInputShape; } interface DecodeFixture { name: string; hex: string; - expect: { ok: true } | { ok: false; code: string }; + expect: { ok: true; hex?: string } | { ok: false; code: string; message: string }; note: string; } -/** - * Tombstone flips that have LANDED (mirror of EXPECTED_TOMBSTONES in - * differential.test.ts, but per-task so P3.5 and P3.7 can land separately). - * A landed task's tombstone-marked fixtures must be regenerated to - * expected-throw; un-landed ones must still encode. - */ -const LANDED_TOMBSTONES = new Set<"P3.5" | "P3.7">(["P3.5", "P3.7"]); +interface FormatFixture { + name: string; + input: { recipe: Recipe } | { hex: string }; + config: "none" | "standard" | "standard+bignum"; + build?: "default" | "bignum"; + expect: { + hex: string; + diagnostic: string; + annotated: string; + flat: string; + summary: string; + hexAnnotated: string; + }; +} + +type DateFixture = { name: string; expect: DateExpect } & ( + { kind: "decode"; hex: string } | { kind: "display"; recipe: Recipe } +); +type DateExpect = + { ok: true; hex: string; display: string } | { ok: false; code: string; message?: string }; + +interface UintFixture { + name: string; + hex: string; + width: 8 | 16 | 32 | 64; + expect: { ok: true; value: string } | { ok: false; code: string }; +} const loadFixtures = (file: string): { count: number; vectors: T[] } => JSON.parse(readFileSync(join(here, "vectors", file), "utf8")) as { @@ -68,8 +105,12 @@ const { count: encodeCount, vectors: encodeVectors } = loadFixtures("encode-vectors.json"); const { count: decodeCount, vectors: decodeVectors } = loadFixtures("decode-vectors.json"); +const { count: formatCount, vectors: formatVectors } = + loadFixtures("format-vectors.json"); +const { count: dateCount, vectors: dateVectors } = loadFixtures("date-vectors.json"); +const { count: uintCount, vectors: uintVectors } = loadFixtures("uint-vectors.json"); -const api = redesignedAdapterFor(src); +const api = currentAdapterFor(src); const sha256 = (hex: string): string => createHash("sha256").update(Buffer.from(hex, "hex")).digest("hex"); @@ -103,8 +144,8 @@ describe("golden encode vectors (frozen wire format)", () => { expect(outcome.hex).toBe(vector.expect.hex); } // Determinism lock: encoder output decodes and re-encodes - // byte-identically - except the pinned bare-Float quirk vectors whose - // output the decoder rejects as non-canonical (frozen behavior). + // byte-identically - unless a fixture deliberately pins a decoder + // rejection of the encoder's own output (none do at present). const back = decodeOutcome(api, hexToBytes(outcome.hex)); if (vector.expect.decodeRejects !== undefined) { expect(back).toEqual({ ok: false, stage: "decode", code: vector.expect.decodeRejects }); @@ -130,14 +171,114 @@ describe("golden decode vectors (accept/reject + error codes)", () => { it(vector.name, () => { const outcome = decodeOutcome(api, hexToBytes(vector.hex)); if (vector.expect.ok) { - // Accepts must re-encode to the exact input bytes. - expect(outcome).toEqual({ ok: true, hex: vector.hex }); + // Accepts re-encode to the exact input bytes, unless the fixture pins + // the re-encoding of the decoded node (whole-valued float heads that + // reduce to integers, as in the reference). + expect(outcome).toEqual({ ok: true, hex: vector.expect.hex ?? vector.hex }); } else { // All fixture rejections are decode-stage rejections by construction // (the generator verifies this); `stage` distinguishes them from // decode-accepted-but-reencode-threw, which would be a divergence. expect(outcome).toEqual({ ok: false, stage: "decode", code: vector.expect.code }); + // The message is the reference's error Display text (the Rust + // harness compares it); pin it here so it cannot drift. + expect(decodeErrorMessage(api, hexToBytes(vector.hex))).toBe(vector.expect.message); + } + }); + } +}); + +describe("golden format vectors (diagnostic and annotated hex, per tags store)", () => { + it("fixture file is self-consistent and in sync with the corpus", async () => { + expect(formatVectors.length).toBe(formatCount); + const { formatCorpus } = await import("./vectors/format-corpus"); + expect(formatVectors.map((v) => v.name)).toEqual(formatCorpus.map((c) => c.name)); + }); + + const storeFor = (config: FormatFixture["config"]): TagsStore => { + const store = new TagsStore(); + if (config === "standard") registerStandardTags(store); + if (config === "standard+bignum") registerStandardTags(store, { bignum: true }); + return store; + }; + + for (const vector of formatVectors) { + it(vector.name, () => { + const value = + "hex" in vector.input + ? decodeCbor(hexToBytes(vector.input.hex)) + : cbor(materialize(vector.input.recipe, api) as CborInput); + const store = storeFor(vector.config); + const tags = vector.config === "none" ? "none" : store; + expect({ + hex: bytesToHex(encodeCbor(value)), + diagnostic: diagnostic(value, { tags }), + annotated: diagnostic(value, { annotate: true, tags }), + flat: diagnostic(value, { flat: true, tags }), + summary: diagnostic(value, { summarize: true, tags }), + hexAnnotated: hexAnnotated(value, { tagsStore: store }), + }).toEqual(vector.expect); + }); + } +}); + +describe("golden date vectors (CborDate decode and display)", () => { + it("fixture file is self-consistent and in sync with the corpus", async () => { + expect(dateVectors.length).toBe(dateCount); + const { dateCorpus } = await import("./vectors/date-corpus"); + expect(dateVectors.map((v) => v.name)).toEqual(dateCorpus.map((c) => c.name)); + }); + + const outcome = (build: () => CborDate): DateExpect => { + try { + const date = build(); + return { ok: true, hex: bytesToHex(encodeCbor(date.toCbor())), display: date.toString() }; + } catch (e) { + const code = api.errorCode(e); + if (code === undefined) throw e; + return { ok: false, code, message: (e as Error).message }; + } + }; + + for (const vector of dateVectors) { + it(vector.name, () => { + const actual = + vector.kind === "decode" + ? outcome(() => CborDate.fromTaggedCbor(decodeCbor(hexToBytes(vector.hex)))) + : outcome(() => materialize(vector.recipe, api) as CborDate); + if (vector.expect.ok || vector.expect.message !== undefined) { + expect(actual).toEqual(vector.expect); + } else { + // Display rows pin the code only. + expect(actual.ok).toBe(false); + expect(!actual.ok && actual.code).toBe(vector.expect.code); + } + }); + } +}); + +describe("golden unsigned-extraction vectors (expectUnsigned with width and wrapNegative)", () => { + it("fixture file is self-consistent and in sync with the corpus", async () => { + expect(uintVectors.length).toBe(uintCount); + const { uintCorpus } = await import("./vectors/uint-corpus"); + expect(uintVectors.map((v) => v.name)).toEqual(uintCorpus.map((c) => c.name)); + }); + + for (const vector of uintVectors) { + it(vector.name, () => { + let actual: UintFixture["expect"]; + try { + const value = expectUnsigned(decodeCbor(hexToBytes(vector.hex)), { + width: vector.width, + wrapNegative: true, + }); + actual = { ok: true, value: String(value) }; + } catch (e) { + const code = api.errorCode(e); + if (code === undefined) throw e; + actual = { ok: false, code }; } + expect(actual).toEqual(vector.expect); }); } }); @@ -171,16 +312,9 @@ describe("golden vector fixture hygiene", () => { } }); - it("tombstone fixtures match their landed/unlanded state", () => { - // Per-task flip: when P3.5 (or P3.7) lands, add it to LANDED_TOMBSTONES - // and regenerate - its marked fixtures must then be expected-throw while - // the other task's fixtures still encode. + it("tombstone fixtures all expect a throw", () => { for (const vector of encodeVectors.filter((v) => v.tombstone)) { - const landed = LANDED_TOMBSTONES.has(vector.tombstone as "P3.5" | "P3.7"); - expect( - vector.expect.ok, - `${vector.name} (${vector.tombstone}) should ${landed ? "throw" : "encode"}`, - ).toBe(!landed); + expect(vector.expect.ok, `${vector.name} (${vector.tombstone}) should throw`).toBe(false); } }); }); diff --git a/tests/golden.test.ts b/tests/golden.test.ts index edb3274..5d44b8d 100644 --- a/tests/golden.test.ts +++ b/tests/golden.test.ts @@ -1,10 +1,9 @@ /** - * Golden wire-format snapshots (REFACTOR_PLAN Phase 0.3). + * Golden wire-format snapshots. * - * Encodes a broad corpus of values and snapshots the resulting hex. Any change - * to the deterministic wire format - the thing the refactor must NOT alter - - * shows up as a snapshot diff. Also asserts each value round-trips - * byte-identically through decode + re-encode. + * Encodes a broad corpus of values and snapshots the resulting hex, so any + * change to the deterministic wire format shows up as a snapshot diff. Also + * asserts each value round-trips byte-identically through decode + re-encode. */ import { cbor, encodeCbor, decodeCbor, CborMap, taggedValue } from "../src"; diff --git a/tests/hex.property.test.ts b/tests/hex.property.test.ts index 84bbac1..807cdb2 100644 --- a/tests/hex.property.test.ts +++ b/tests/hex.property.test.ts @@ -1,15 +1,13 @@ /** - * Hex front-door property tests (P3.6 gate, run at P4.1). + * Hex conversion property tests. * - * Three obligations from the plan: - * 1. old-vs-new byte equality: for every hex string the PRE-redesign - * `hexToBytes` decoded meaningfully (valid even-length hex), the new - * validating implementation produces bit-identical bytes - compared - * directly against the frozen baseline bundle's export; + * 1. Baseline byte equality: for valid even-length hex, the validating + * `hexToBytes` produces the same bytes as the baseline bundle's + * non-validating one; * 2. round-trips: bytes → hex → bytes and valid hex → bytes → hex are * lossless (case-normalized); - * 3. malformed hex (odd length, non-hex characters) now throws CborError - * code "Custom" instead of silently fabricating bytes. + * 3. malformed hex (odd length, non-hex characters) throws CborError code + * "Custom", where the baseline fabricated bytes. */ import fc from "fast-check"; @@ -23,8 +21,8 @@ const validHexArb = fc .map((idxs) => idxs.map((i) => hexChars[i]).join("")) .filter((s) => s.length % 2 === 0); -describe("hex property tests (old vs new, P3.6)", () => { - it("valid hex decodes bit-identically to the pre-redesign implementation", () => { +describe("hex property tests (baseline vs working tree)", () => { + it("valid hex decodes bit-identically to the baseline implementation", () => { fc.assert( fc.property(validHexArb, (hex) => { const oldBytes = baseline.hexToBytes(hex); diff --git a/tests/map.test.ts b/tests/map.test.ts new file mode 100644 index 0000000..34f6873 --- /dev/null +++ b/tests/map.test.ts @@ -0,0 +1,88 @@ +/** + * `CborMap` positional access - the lockstep walk `cborEquals` and the encoder + * use instead of materializing `entriesArray`. + */ + +import { cbor, encodeCbor, cborEquals } from "../src/cbor"; +import { CborMap } from "../src/map"; +import { bytesToHex } from "../src/hex"; +import { lexicographicallyCompareBytes } from "../src/stdlib"; + +describe("CborMap positional access (entryAt / encodedKeyAt)", () => { + const build = (): CborMap => { + const m = new CborMap(); + m.set("z", 1); + m.set(10, "ten"); + m.set("é", true); // non-NFC text key + m.set([1, 2], null); + m.set(-1, 0.5); + return m; + }; + + it("walks the same entries as entriesArray, in canonical order", () => { + const m = build(); + const entries = m.entriesArray; + expect(m.size).toBe(5); + expect(entries).toHaveLength(m.size); + for (let i = 0; i < m.size; i++) { + expect(cborEquals(m.entryAt(i).key, entries[i].key)).toBe(true); + expect(cborEquals(m.entryAt(i).value, entries[i].value)).toBe(true); + } + for (let i = 1; i < m.size; i++) { + expect(lexicographicallyCompareBytes(m.encodedKeyAt(i - 1), m.encodedKeyAt(i))).toBeLessThan( + 0, + ); + } + }); + + it("stores the encoded key bytes each entry is sorted by", () => { + const m = build(); + for (let i = 0; i < m.size; i++) { + expect(bytesToHex(m.encodedKeyAt(i))).toBe(bytesToHex(encodeCbor(m.entryAt(i).key))); + } + }); + + it("encodes a non-NFC text key through its stored (NFC) bytes", () => { + // The node keeps the decomposed string; the stored key bytes, and so the + // wire, carry the composed form - the same bytes re-encoding would give. + const m = new CborMap(); + m.set("é", true); + expect(m.entryAt(0).key.type).toBe(3); + expect(m.entryAt(0).key.value).toBe("é"); + expect(bytesToHex(m.encodedKeyAt(0))).toBe("62c3a9"); + expect(bytesToHex(encodeCbor(cbor(m)))).toBe("a162c3a9f5"); + }); + + it("encodes to the same bytes regardless of insertion order", () => { + const a = build(); + const b = new CborMap(); + b.set(-1, 0.5); + b.set([1, 2], null); + b.set("é", true); + b.set(10, "ten"); + b.set("z", 1); + expect(bytesToHex(encodeCbor(cbor(a)))).toBe(bytesToHex(encodeCbor(cbor(b)))); + expect(cborEquals(cbor(a), cbor(b))).toBe(true); + }); + + it("a replaced value keeps the entry's position and key bytes", () => { + const m = build(); + const before = m.entriesArray.map((e) => e.key); + m.set(10, "TEN"); + expect(m.size).toBe(5); + for (let i = 0; i < m.size; i++) { + expect(cborEquals(m.entryAt(i).key, before[i])).toBe(true); + } + expect(m.get(10)?.value).toBe("TEN"); + }); + + it("structural equality stops at the first differing entry", () => { + const a = build(); + const b = build(); + b.set("z", 2); + expect(cborEquals(cbor(a), cbor(b))).toBe(false); + const c = build(); + c.delete(-1); + expect(cborEquals(cbor(a), cbor(c))).toBe(false); // size differs + }); +}); diff --git a/tests/roundtrip.property.test.ts b/tests/roundtrip.property.test.ts index 03f4f73..c17ee38 100644 --- a/tests/roundtrip.property.test.ts +++ b/tests/roundtrip.property.test.ts @@ -1,10 +1,9 @@ /** - * Round-trip stability property test (REFACTOR_PLAN Phase 0.4). + * Round-trip stability property test. * * For any dCBOR-representable value, re-encoding the decoded bytes must produce - * byte-identical output: encode(decode(encode(v))) === encode(v). This is the - * strongest safety net for the encoder/decoder rewrite in later phases - it - * exercises far more inputs than the hand-written vectors. + * byte-identical output: encode(decode(encode(v))) === encode(v). It exercises + * far more inputs than the hand-written vectors. * * Strings are restricted to printable ASCII so generated inputs are always * valid UTF-8 and in NFC (dedicated tests cover the UTF-8/NFC rejection paths); diff --git a/tests/rust-validation/Cargo.lock b/tests/rust-validation/Cargo.lock index e43edf1..2357fe9 100644 --- a/tests/rust-validation/Cargo.lock +++ b/tests/rust-validation/Cargo.lock @@ -111,6 +111,7 @@ dependencies = [ name = "dcbor-vector-validation" version = "0.1.0" dependencies = [ + "chrono", "dcbor", "hex", "num-bigint", diff --git a/tests/rust-validation/Cargo.toml b/tests/rust-validation/Cargo.toml index 8edd161..f903019 100644 --- a/tests/rust-validation/Cargo.toml +++ b/tests/rust-validation/Cargo.toml @@ -7,12 +7,22 @@ description = "Cross-validates @blockchaincommons/dcbor golden wire vectors agai [dependencies] # The reference implementation @blockchaincommons/dcbor is a port of. Keep the pin in sync -# with the version the vectors were validated against. -dcbor = { version = "=0.25.2", features = ["num-bigint"] } -num-bigint = "0.4.6" +# with the version the vectors were validated against. The default build is what every +# Rust consumer in the Blockchain Commons stack uses (no `num-bigint`); the `bignum` +# feature below enables the reference's optional bignum support for a second run. +dcbor = { version = "=0.25.2" } +# chrono is dcbor's date backend; used here only to probe `timestamp_opt` +# for the inputs `Date::from_timestamp` panics on (see the "date" arm). +chrono = "0.4" +num-bigint = { version = "0.4.6", optional = true } serde_json = "1" sha2 = "0.10" hex = "0.4" +[features] +# `cargo run --release --features bignum -- ../vectors`: the reference with +# `num-bigint`, which names tags 2/3 and materializes the biguint/bignum recipes. +bignum = ["dcbor/num-bigint", "dep:num-bigint"] + [profile.release] opt-level = 2 diff --git a/tests/rust-validation/README.md b/tests/rust-validation/README.md index 2a3ac7c..e3335ca 100644 --- a/tests/rust-validation/README.md +++ b/tests/rust-validation/README.md @@ -1,44 +1,58 @@ -# Rust reference cross-validation (P1.1 hardening) +# Rust reference cross-validation -Validates the committed golden wire vectors (`tests/vectors/*.json`) against -the Rust reference implementation this library is a port of +Validates the committed golden vectors (`tests/vectors/*.json`) against the +Rust reference implementation this library is a port of ([bc-dcbor-rust](https://github.com/BlockchainCommons/bc-dcbor-rust), crate `dcbor`, pinned `=0.25.2`). +The harness runs twice: once against the default build every Rust consumer +in the Blockchain Commons stack uses, once with dcbor's `num-bigint` +feature, which names tags 2/3 and can materialize the bignum recipes. + ```sh cd tests/rust-validation cargo run --release -- ../vectors +cargo run --release --features bignum -- ../vectors ``` -Exit code 0 iff every vector either matches Rust byte-for-byte / -code-for-code, or falls into one of the small, explicitly-named classes: - -- **skipped (3)** - JS-only inputs with no Rust analog: `Symbol`, function, - malformed bare Cbor node. -- **emulated-throw (6)** - TS *input guards* the harness mirrors because - Rust's typed API cannot express the input: `cbor(bigint)` range guard - (Rust has no `i128 → CBOR`; its own tests use `CBORCase` directly), - `biguintToCbor(<0)`, and two `Date::from_string` failures (those two are - in fact validated through Rust's own parser). -- **expected-divergence (6)** - documented, allowlisted TS↔Rust behavioral - differences: - - 5 decode vectors: byte-string/text lengths ≥ 2^53 → TS `OutOfRange` - (JS bigint length narrowing) vs Rust `Underrun` (usize length, body - bounds check). - - 1 encode vector: `CborDate.fromTimestamp(NaN)` → TS throws - `InvalidDate`; Rust `Date::from_timestamp` saturating-casts to epoch. - -Everything else - including the bare-Float-node quirk vectors (`fround` -negative-reduction collisions, f32-exact wholes ≥ 2^32 staying `0xfa` -floats) and all 177 decode rejections with exact error codes - matches the -Rust reference exactly, confirming the frozen TS behavior is genuine Rust -parity and not porting artifacts. - -Every divergence class is documented in detail in -[`RUST_DIVERGENCES.md`](../../RUST_DIVERGENCES.md) at the repo root; the -`expected_divergences()` allowlist in `src/main.rs` is its machine-readable -twin - keep them in sync. - -Re-run this after any fixture regeneration (and at the P4.1 proof -re-baseline). It is not wired into the JS CI job because it needs a Rust -toolchain; treat it as a mandatory manual gate at phase boundaries. +Both runs are a CI job (`.github/workflows/ci.yml`, `rust-validation`). + +## What is compared + +| file | Rust side | compared | +|---|---|---| +| `encode-vectors.json` | the recipe materialized with the `dcbor` API, `to_cbor_data()` | bytes (or digest + length), or the error code | +| `decode-vectors.json` | `CBOR::try_from_data`, then `to_cbor_data()` | re-encoded bytes (the fixture may pin a different re-encoding for whole-valued float heads), or the error code **and** its `Display` message | +| `format-vectors.json` | `diagnostic_opt` (plain, annotated, flat, summarized) and `hex_opt` under `TagsStoreOpt::None` or a fresh store after `register_tags_in` | all five strings | +| `date-vectors.json` | `Date::from_tagged_cbor`, `Date::from_timestamp`, `Date::from_string`, then `to_cbor_data()` and `Display` | bytes + display, or the error code (and message for decode rows) | +| `uint-vectors.json` | `u8`/`u16`/`u32`/`u64::try_from(CBOR)` | the value, or the error code | + +Exit code 0 iff every vector either matches, or falls into one of the +explicitly named classes: + +- **reference-throw** - the reference itself rejects the input + (`Date::from_string` on an unparsable string) with the fixture's code. +- **emulated-throw** - a TypeScript guard the harness mirrors because the + reference cannot express the input (`cbor(bigint)` outside the CBOR integer + range, a negative `biguintToCbor`) or would panic on it + (`Date::from_timestamp` on ±Infinity or on whole seconds outside chrono's + range, probed with chrono's own `timestamp_opt` before the call). +- **skipped** - JS-only inputs with no Rust analog (`Symbol`, function, + malformed bare node), the rejected `{tag, value}` literal and + `taggedCbor()`-only shapes (fixtures marked `tombstone`), rows pinned to + the other build, and the bignum recipes in the default build. +- **expected-divergence** - a documented TS↔Rust difference allowlisted by + vector name in `expected_divergences()` in `src/main.rs`. The list is + empty: every recorded divergence is closed. The mechanism stays so a + deliberate future divergence is recorded in + [`RUST_DIVERGENCES.md`](../../RUST_DIVERGENCES.md) and allowlisted here in + the same change. + +Everything else - the whole-valued float heads that decode to integers, the +bare-Float-node quirks, every decode rejection with its exact code and +message, every diagnostic and annotated-hex rendering, every date display +including leap seconds - matches the reference exactly. + +Re-run both builds after any fixture regeneration (`bun run vectors:generate`) +and update the result lines in `RUST_DIVERGENCES.md`. When bumping the +`dcbor` pin, re-run both builds and update the header there too. diff --git a/tests/rust-validation/src/main.rs b/tests/rust-validation/src/main.rs index b56769d..09488e2 100644 --- a/tests/rust-validation/src/main.rs +++ b/tests/rust-validation/src/main.rs @@ -1,55 +1,74 @@ -//! Cross-validates the @blockchaincommons/dcbor golden wire vectors against the Rust +//! Cross-validates the @blockchaincommons/dcbor golden vectors against the Rust //! reference implementation (`dcbor` crate, bc-dcbor-rust). //! -//! Usage: cargo run --release -- +//! Usage (run both; the reference's `num-bigint` feature changes how tags 2 +//! and 3 are named and which recipes it can materialize): //! -//! Reads `encode-vectors.json` and `decode-vectors.json`, materializes each -//! recipe with the Rust API, encodes/decodes, and compares against the -//! committed expectations. Every vector is classified as: +//! cargo run --release -- +//! cargo run --release --features bignum -- //! -//! match - Rust produces exactly the fixture outcome -//! emulated - the recipe describes a JS-side *input guard* (e.g. bigint -//! out-of-CBOR-range throwing OutOfRange) that the harness -//! re-implements because Rust's typed API makes the input -//! inexpressible; the vector validates by construction -//! skipped - JS-only input shape with no Rust analog (Symbol, function, -//! malformed bare node) -//! expected-divergence - a known, documented TS↔Rust behavioral difference -//! (allowlisted by vector name below) -//! MISMATCH - anything else; fails the run +//! Reads the five fixture files and compares the reference's outcome with the +//! committed expectation: +//! +//! encode-vectors.json recipe -> bytes (or a CborError code) +//! decode-vectors.json bytes -> re-encoded bytes, or a code AND message +//! format-vectors.json value + tags store -> diagnostic (plain, annotated, +//! flat, summary) and annotated hex +//! date-vectors.json tag-1 bytes / date recipe -> bytes + Display, or code +//! uint-vectors.json bytes -> u8/u16/u32/u64::try_from +//! +//! Every vector is classified as: +//! +//! match - the reference produces exactly the fixture outcome +//! reference-throw - the reference itself rejects the input +//! (`Date::from_string`), with the fixture's code +//! emulated-throw - a TS guard the harness mirrors because the reference +//! cannot express the input (`cbor(bigint)` outside the +//! CBOR integer range, a negative `biguintToCbor`) or +//! would panic on it (`Date::from_timestamp` on ±Infinity +//! or whole seconds outside chrono's range - probed with +//! chrono's own `timestamp_opt` before the call) +//! skipped - JS-only input shape (Symbol, function, malformed bare +//! node), a tombstoned recipe, a row pinned to the other +//! build, or a bignum recipe in the default build +//! expected-divergence - a documented TS<->Rust difference allowlisted by +//! vector name in `expected_divergences()` (empty today) +//! MISMATCH - anything else; fails the run //! //! Exit code 0 iff there are no MISMATCHes. use std::collections::BTreeMap; use std::process::ExitCode; +use chrono::{LocalResult, TimeZone, Utc}; use dcbor::prelude::*; -use dcbor::Simple; -use num_bigint::{BigInt, BigUint}; +use dcbor::{register_tags_in, DiagFormatOpts, HexFormatOpts, Simple, TagsStoreOpt}; use serde_json::Value; use sha2::{Digest, Sha256}; +/// The build this binary was compiled as; format rows may be pinned to one. +const BUILD: &str = if cfg!(feature = "bignum") { "bignum" } else { "default" }; + /// Known, documented TS↔Rust divergences: vector name -> (expected Rust /// outcome, reason). Anything diverging outside this list is a MISMATCH. +/// +/// Empty: every recorded divergence is closed. The mechanism stays so a +/// future, deliberate divergence is allowlisted by name here and recorded in +/// RUST_DIVERGENCES.md rather than hidden. fn expected_divergences() -> BTreeMap<&'static str, (&'static str, &'static str)> { - BTreeMap::from([ - // TS CborDate.fromTimestamp throws InvalidDate for non-finite input; - // Rust from_timestamp saturating-casts (NaN -> epoch 0). - ( - "date/non-finite-throws", - ("bytes c100", "TS guards non-finite timestamps; Rust saturates"), - ), - ]) + BTreeMap::new() } enum Materialized { Value(CBOR), + /// The reference rejected the input itself (a real `dcbor::Error`). + ReferenceThrow(&'static str), /// TS-side input guard reproduced by the harness (see module docs). EmulatedThrow(&'static str), Skip(&'static str), } -use Materialized::{EmulatedThrow, Skip, Value as Mat}; +use Materialized::{EmulatedThrow, ReferenceThrow, Skip, Value as Mat}; fn parse_f64(v: &str) -> f64 { match v { @@ -83,6 +102,21 @@ fn cycle_bytes(start: u64, count: u64) -> Vec { (0..count).map(|i| ((start + i) & 0xff) as u8).collect() } +/// `Date::from_timestamp` is `Utc.timestamp_opt(trunc as i64, (fract * 1e9) +/// as u32).unwrap()`: it panics where chrono has no such instant (±inf +/// saturate to the i64 bounds; whole seconds outside ±262143 years), and TS +/// throws InvalidDate there. Probe chrono with the reference's own +/// arithmetic and emulate the throw. NaN is NOT guarded: `trunc() as i64` +/// saturates it to 0, the epoch, on both sides. +fn date_from_timestamp(secs: f64) -> Result { + let whole = secs.trunc() as i64; + let nsecs = (secs.fract() * 1_000_000_000.0) as u32; + if matches!(Utc.timestamp_opt(whole, nsecs), LocalResult::None) { + return Err(EmulatedThrow("InvalidDate")); + } + Ok(Date::from_timestamp(secs)) +} + /// SameValueZero-style dedup key for the jsset emulation (JS Set semantics: /// numbers by value with NaN==NaN and +0==-0; bigints/strings/bools/null by /// value but distinct across kinds; objects by identity - never deduped). @@ -213,58 +247,51 @@ fn materialize(recipe: &Value) -> Materialized { other => other, } } - // JS {tag, value} sniffing produces exactly to_tagged_value bytes; - // the tag recipe is a number (or a string that Number()-coerces). - "tagobjlit" => { - let tag_recipe = &recipe["tag"]; - let tag = match tag_recipe["k"].as_str().unwrap() { - "n" => parse_f64(tag_recipe["v"].as_str().unwrap()) as u64, - "s" => parse_f64(tag_recipe["v"].as_str().unwrap()) as u64, - _ => return Skip("tagobjlit with non-numeric tag recipe"), - }; - match materialize(&recipe["content"]) { - Mat(c) => Mat(CBOR::to_tagged_value(tag, c)), - other => other, - } - } - "date" => { - let secs = parse_f64(recipe["seconds"].as_str().unwrap()); - if !secs.is_finite() { - // TS throws InvalidDate; Rust saturates. Materialize the - // Rust behavior and let the divergence allowlist judge it. - return Mat(Date::from_timestamp(0.0).into()); - } - Mat(Date::from_timestamp(secs).into()) - } + // Removed input shapes (a `{tag, value}` object literal, a + // `taggedCbor`-only object): their fixtures expect a TS directive + // error and are skipped before materialization (see run_encode). + "tagobjlit" | "taggedproto" => Skip("tombstoned JS-only input shape"), + "date" => match date_from_timestamp(parse_f64(recipe["seconds"].as_str().unwrap())) { + Ok(d) => Mat(d.into()), + Err(m) => m, + }, "datestr" => match Date::from_string(recipe["v"].as_str().unwrap()) { Ok(d) => Mat(d.into()), - Err(_) => EmulatedThrow("InvalidDate"), + Err(_) => ReferenceThrow("InvalidDate"), }, "bytestring" => Mat(CBOR::to_byte_string( hex::decode(recipe["hex"].as_str().unwrap()).unwrap(), )), "biguint" => { let v = recipe["v"].as_str().unwrap(); - if let Some(stripped) = v.strip_prefix('-') { - let _ = stripped; + if v.starts_with('-') { return EmulatedThrow("OutOfRange"); // TS biguintToCbor(<0) } - Mat(CBOR::from(v.parse::().unwrap())) + #[cfg(feature = "bignum")] + { + Mat(CBOR::from(v.parse::().unwrap())) + } + #[cfg(not(feature = "bignum"))] + { + Skip("needs num-bigint") + } } - "bignum" => Mat(CBOR::from( - recipe["v"].as_str().unwrap().parse::().unwrap(), - )), - // Protocol wrappers: byte-equivalent to their underlying values. - "tocbor" => materialize(&recipe["inner"]), - "taggedproto" => { - let tag = tag_from_str(recipe["tag"].as_str().unwrap()); - match materialize(&recipe["inner"]) { - Mat(c) => Mat(CBOR::to_tagged_value(tag, c)), - other => other, + "bignum" => { + #[cfg(feature = "bignum")] + { + Mat(CBOR::from( + recipe["v"].as_str().unwrap().parse::().unwrap(), + )) + } + #[cfg(not(feature = "bignum"))] + { + Skip("needs num-bigint") } } - // Post-P3.7 dispatch precedence: toCbor() wins over taggedCbor(), - // so bothproto encodes as the toCbor side's marker array. + // Protocol wrappers: byte-equivalent to their underlying values. + "tocbor" => materialize(&recipe["inner"]), + // Dispatch precedence: toCbor() wins over taggedCbor(), so bothproto + // encodes as the toCbor side's marker array. "bothproto" => match materialize(&recipe["inner"]) { Mat(c) => Mat(vec![CBOR::from("toCbor-won"), c].into()), other => other, @@ -287,6 +314,9 @@ fn materialize(recipe: &Value) -> Materialized { ))) .into()), "rawuint" => Mat(CBORCase::Unsigned(recipe["v"].as_str().unwrap().parse().unwrap()).into()), + // Bare Text node: the string is stored verbatim and NFC-normalized by + // `cbor_data` at encode time (cbor.rs), exactly like the TS node. + "rawtext" => Mat(CBORCase::Text(recipe["v"].as_str().unwrap().to_string()).into()), "rawnegmag" => { Mat(CBORCase::Negative(recipe["v"].as_str().unwrap().parse().unwrap()).into()) } @@ -324,9 +354,16 @@ fn sha256_hex(bytes: &[u8]) -> String { hex::encode(Sha256::digest(bytes)) } +#[derive(Clone, Copy)] +enum ThrowClass { + Reference, + Emulated, +} + #[derive(Default)] struct Tally { matched: usize, + reference_throw: usize, emulated: usize, skipped: usize, expected_divergence: usize, @@ -337,13 +374,47 @@ impl Tally { fn mismatch(&mut self, name: &str, detail: String) { self.mismatches.push(format!("{name}: {detail}")); } + + /// Compare a Rust outcome string with the fixture's, honouring the + /// allowlist; `throw_class` says which throw tally a matching throw feeds. + fn judge( + &mut self, + name: &str, + rust: &str, + ts: &str, + throw_class: ThrowClass, + divergences: &BTreeMap<&str, (&str, &str)>, + ) { + if rust == ts { + if rust.starts_with("throw") { + match throw_class { + ThrowClass::Reference => self.reference_throw += 1, + ThrowClass::Emulated => self.emulated += 1, + } + } else { + self.matched += 1; + } + } else if let Some((allowed, _why)) = divergences.get(name) { + if rust == *allowed { + self.expected_divergence += 1; + } else { + self.mismatch( + name, + format!("divergence allowlisted as '{allowed}' but Rust gave '{rust}'"), + ); + } + } else { + self.mismatch(name, format!("TS {ts} != Rust {rust}")); + } + } } -/// The Rust-side outcome of a vector, as a comparable string. -fn outcome_string(m: Materialized) -> Result { +/// The Rust-side outcome of a recipe, as a comparable string. +fn outcome_string(m: Materialized) -> Result<(String, ThrowClass), &'static str> { match m { - Mat(c) => Ok(format!("bytes {}", hex::encode(c.to_cbor_data()))), - EmulatedThrow(code) => Ok(format!("throw {code}")), + Mat(c) => Ok((format!("bytes {}", hex::encode(c.to_cbor_data())), ThrowClass::Reference)), + ReferenceThrow(code) => Ok((format!("throw {code}"), ThrowClass::Reference)), + EmulatedThrow(code) => Ok((format!("throw {code}"), ThrowClass::Emulated)), Skip(reason) => Err(reason), } } @@ -353,16 +424,15 @@ fn run_encode(vectors: &[Value], tally: &mut Tally, divergences: &BTreeMap<&str, let name = vector["name"].as_str().unwrap(); let expect = &vector["expect"]; - // Post-P3 tombstones: fixtures marked with a plan-task tombstone that - // now expect a throw exercise TS-only directive errors (the {tag, - // value} sniffing removal and the taggedCbor auto-wrap removal) - - // there is no Rust analog to compare. + // Tombstone fixtures that expect a throw exercise TS-only directive + // errors (a `{tag, value}` object literal, a `taggedCbor`-only + // object) - there is no Rust analog to compare. if vector["tombstone"].is_string() && expect["ok"].as_bool() == Some(false) { tally.skipped += 1; continue; } - let rust_outcome = match outcome_string(materialize(&vector["recipe"])) { + let (rust_outcome, throw_class) = match outcome_string(materialize(&vector["recipe"])) { Ok(o) => o, Err(_reason) => { tally.skipped += 1; @@ -393,24 +463,7 @@ fn run_encode(vectors: &[Value], tally: &mut Tally, divergences: &BTreeMap<&str, format!("throw {}", expect["code"].as_str().unwrap()) }; - if rust_outcome == ts_outcome { - if rust_outcome.starts_with("throw") { - tally.emulated += 1; // throws are TS input guards the harness mirrors - } else { - tally.matched += 1; - } - } else if let Some((allowed, _why)) = divergences.get(name) { - if rust_outcome == *allowed { - tally.expected_divergence += 1; - } else { - tally.mismatch( - name, - format!("divergence allowlisted as '{allowed}' but Rust gave '{rust_outcome}'"), - ); - } - } else { - tally.mismatch(name, format!("TS {ts_outcome} != Rust {rust_outcome}")); - } + tally.judge(name, &rust_outcome, &ts_outcome, throw_class, divergences); } } @@ -422,29 +475,211 @@ fn run_decode(vectors: &[Value], tally: &mut Tally, divergences: &BTreeMap<&str, let rust_outcome = match CBOR::try_from_data(&bytes) { Ok(c) => format!("bytes {}", hex::encode(c.to_cbor_data())), - Err(e) => format!("throw {}", error_code(&e)), + // Rejections compare by code AND by the error's Display text: the + // message is part of the contract the port keeps. + Err(e) => format!("throw {} / {}", error_code(&e), e), }; let ts_outcome = if expect["ok"].as_bool().unwrap() { - format!("bytes {}", vector["hex"].as_str().unwrap()) + // An accept fixture may pin a re-encoding that differs from the + // input (whole-valued f32/f64 heads decode to integer nodes); + // otherwise the accept must round-trip byte-identically. + let hex = expect["hex"] + .as_str() + .unwrap_or_else(|| vector["hex"].as_str().unwrap()); + format!("bytes {hex}") } else { - format!("throw {}", expect["code"].as_str().unwrap()) + format!( + "throw {} / {}", + expect["code"].as_str().unwrap(), + expect["message"].as_str().unwrap_or("") + ) }; - if rust_outcome == ts_outcome { - tally.matched += 1; - } else if let Some((allowed, _why)) = divergences.get(name) { - if rust_outcome == *allowed { - tally.expected_divergence += 1; - } else { - tally.mismatch( - name, - format!("divergence allowlisted as '{allowed}' but Rust gave '{rust_outcome}'"), - ); + tally.judge(name, &rust_outcome, &ts_outcome, ThrowClass::Reference, divergences); + } +} + +/// The tags store a format row asks for, as the reference's `TagsStoreOpt`. +fn tags_store_for(config: &str) -> Option { + match config { + "none" => None, + "standard" | "standard+bignum" => { + // `register_tags_in` registers whatever this build supports: the + // date tag, plus tags 2/3 under `num-bigint`. Rows are pinned to + // the build whose store they describe. + let mut store = TagsStore::default(); + register_tags_in(&mut store); + Some(store) + } + other => panic!("unknown format config: {other}"), + } +} + +fn run_format(vectors: &[Value], tally: &mut Tally, divergences: &BTreeMap<&str, (&str, &str)>) { + for vector in vectors { + let name = vector["name"].as_str().unwrap(); + if let Some(build) = vector["build"].as_str() { + if build != BUILD { + tally.skipped += 1; + continue; + } + } + let input = &vector["input"]; + let value = if let Some(h) = input["hex"].as_str() { + match CBOR::try_from_data(&hex::decode(h).unwrap()) { + Ok(c) => c, + Err(e) => { + tally.mismatch(name, format!("input bytes do not decode: {e}")); + continue; + } } } else { - tally.mismatch(name, format!("TS {ts_outcome} != Rust {rust_outcome}")); + match materialize(&input["recipe"]) { + Mat(c) => c, + Skip(_) => { + tally.skipped += 1; + continue; + } + ReferenceThrow(code) | EmulatedThrow(code) => { + tally.mismatch(name, format!("input recipe throws {code}")); + continue; + } + } + }; + let store = tags_store_for(vector["config"].as_str().unwrap()); + let opt = || match &store { + None => TagsStoreOpt::None, + Some(s) => TagsStoreOpt::Custom(s), + }; + let expect = &vector["expect"]; + let rendered: [(&str, String); 6] = [ + ("hex", value.hex()), + ("diagnostic", value.diagnostic_opt(&DiagFormatOpts::default().tags(opt()))), + ( + "annotated", + value.diagnostic_opt(&DiagFormatOpts::default().annotate(true).tags(opt())), + ), + ("flat", value.diagnostic_opt(&DiagFormatOpts::default().flat(true).tags(opt()))), + ( + "summary", + value.diagnostic_opt(&DiagFormatOpts::default().summarize(true).tags(opt())), + ), + ( + "hexAnnotated", + value.hex_opt(&HexFormatOpts::default().annotate(true).context(opt())), + ), + ]; + let mut rust = String::new(); + let mut ts = String::new(); + for (field, actual) in &rendered { + let expected = expect[*field].as_str().unwrap_or(""); + rust.push_str(&format!("{field}={actual:?}\n")); + ts.push_str(&format!("{field}={expected:?}\n")); + } + tally.judge(name, &rust, &ts, ThrowClass::Reference, divergences); + } +} + +/// Whether `Date::from_tagged_cbor` on this value would panic inside +/// `from_timestamp` (±inf, whole seconds outside chrono's range). Anything +/// else runs the real decoder, so the real error (WrongType, WrongTag, +/// OutOfRange) is reported with its message. +fn date_decode_panics(cbor: &CBOR) -> bool { + if let CBORCase::Tagged(tag, item) = cbor.as_case() { + if tag.value() == 1 { + if let Ok(secs) = f64::try_from(item.clone()) { + return date_from_timestamp(secs).is_err(); + } } } + false +} + +fn run_date(vectors: &[Value], tally: &mut Tally, divergences: &BTreeMap<&str, (&str, &str)>) { + for vector in vectors { + let name = vector["name"].as_str().unwrap(); + let expect = &vector["expect"]; + let kind = vector["kind"].as_str().unwrap(); + + // Rust outcome: "bytes display " or "throw [ / ]". + let (rust_outcome, throw_class) = if kind == "decode" { + let bytes = hex::decode(vector["hex"].as_str().unwrap()).unwrap(); + match CBOR::try_from_data(&bytes) { + Err(e) => (format!("throw {} / {}", error_code(&e), e), ThrowClass::Reference), + Ok(cbor) if date_decode_panics(&cbor) => { + ("throw InvalidDate".to_string(), ThrowClass::Emulated) + } + Ok(cbor) => match Date::from_tagged_cbor(cbor) { + Ok(d) => ( + format!("bytes {} display {}", hex::encode(d.to_cbor_data()), d), + ThrowClass::Reference, + ), + Err(e) => (format!("throw {} / {}", error_code(&e), e), ThrowClass::Reference), + }, + } + } else { + let recipe = &vector["recipe"]; + let date = match recipe["k"].as_str().unwrap() { + "date" => date_from_timestamp(parse_f64(recipe["seconds"].as_str().unwrap())), + "datestr" => Date::from_string(recipe["v"].as_str().unwrap()) + .map_err(|_| ReferenceThrow("InvalidDate")), + other => panic!("date display recipe must be date/datestr, got {other}"), + }; + match date { + Ok(d) => ( + format!("bytes {} display {}", hex::encode(d.to_cbor_data()), d), + ThrowClass::Reference, + ), + Err(EmulatedThrow(code)) => (format!("throw {code}"), ThrowClass::Emulated), + Err(ReferenceThrow(code)) => (format!("throw {code}"), ThrowClass::Reference), + Err(_) => unreachable!(), + } + }; + + let ts_outcome = if expect["ok"].as_bool().unwrap() { + format!( + "bytes {} display {}", + expect["hex"].as_str().unwrap(), + expect["display"].as_str().unwrap() + ) + } else if let (Some(message), ThrowClass::Reference) = + (expect["message"].as_str(), throw_class) + { + format!("throw {} / {}", expect["code"].as_str().unwrap(), message) + } else { + // The reference panics here (emulated), or the row pins the code only. + format!("throw {}", expect["code"].as_str().unwrap()) + }; + + tally.judge(name, &rust_outcome, &ts_outcome, throw_class, divergences); + } +} + +fn run_uint(vectors: &[Value], tally: &mut Tally, divergences: &BTreeMap<&str, (&str, &str)>) { + for vector in vectors { + let name = vector["name"].as_str().unwrap(); + let bytes = hex::decode(vector["hex"].as_str().unwrap()).unwrap(); + let cbor = CBOR::try_from_data(&bytes).expect("uint vector must decode"); + let width = vector["width"].as_u64().unwrap(); + let result: Result = match width { + 8 => u8::try_from(cbor).map(|v| v.to_string()), + 16 => u16::try_from(cbor).map(|v| v.to_string()), + 32 => u32::try_from(cbor).map(|v| v.to_string()), + 64 => u64::try_from(cbor).map(|v| v.to_string()), + other => panic!("unsupported width {other}"), + }; + let rust_outcome = match result { + Ok(v) => format!("value {v}"), + Err(e) => format!("throw {}", error_code(&e)), + }; + let expect = &vector["expect"]; + let ts_outcome = if expect["ok"].as_bool().unwrap() { + format!("value {}", expect["value"].as_str().unwrap()) + } else { + format!("throw {}", expect["code"].as_str().unwrap()) + }; + tally.judge(name, &rust_outcome, &ts_outcome, ThrowClass::Reference, divergences); + } } fn main() -> ExitCode { @@ -466,33 +701,73 @@ fn main() -> ExitCode { let divergences = expected_divergences(); let mut encode_tally = Tally::default(); let mut decode_tally = Tally::default(); + let mut format_tally = Tally::default(); + let mut date_tally = Tally::default(); + let mut uint_tally = Tally::default(); let encode_vectors = load("encode-vectors.json"); let decode_vectors = load("decode-vectors.json"); + let format_vectors = load("format-vectors.json"); + let date_vectors = load("date-vectors.json"); + let uint_vectors = load("uint-vectors.json"); run_encode(&encode_vectors, &mut encode_tally, &divergences); run_decode(&decode_vectors, &mut decode_tally, &divergences); + run_format(&format_vectors, &mut format_tally, &divergences); + run_date(&date_vectors, &mut date_tally, &divergences); + run_uint(&uint_vectors, &mut uint_tally, &divergences); + println!("build: {BUILD}"); println!( - "encode: {} vectors - {} match, {} emulated-throw, {} skipped (JS-only), {} expected-divergence, {} MISMATCH", + "encode: {} vectors - {} match, {} reference-throw, {} emulated-throw, {} skipped, {} expected-divergence, {} MISMATCH", encode_vectors.len(), encode_tally.matched, + encode_tally.reference_throw, encode_tally.emulated, encode_tally.skipped, encode_tally.expected_divergence, encode_tally.mismatches.len() ); println!( - "decode: {} vectors - {} match, {} expected-divergence, {} MISMATCH", + "decode: {} vectors - {} match, {} reference-throw, {} expected-divergence, {} MISMATCH", decode_vectors.len(), decode_tally.matched, + decode_tally.reference_throw, decode_tally.expected_divergence, decode_tally.mismatches.len() ); + println!( + "format: {} vectors - {} match, {} skipped, {} expected-divergence, {} MISMATCH", + format_vectors.len(), + format_tally.matched, + format_tally.skipped, + format_tally.expected_divergence, + format_tally.mismatches.len() + ); + println!( + "date: {} vectors - {} match, {} reference-throw, {} emulated-throw, {} expected-divergence, {} MISMATCH", + date_vectors.len(), + date_tally.matched, + date_tally.reference_throw, + date_tally.emulated, + date_tally.expected_divergence, + date_tally.mismatches.len() + ); + println!( + "uint: {} vectors - {} match, {} reference-throw, {} expected-divergence, {} MISMATCH", + uint_vectors.len(), + uint_tally.matched, + uint_tally.reference_throw, + uint_tally.expected_divergence, + uint_tally.mismatches.len() + ); let all: Vec<&String> = encode_tally .mismatches .iter() .chain(decode_tally.mismatches.iter()) + .chain(format_tally.mismatches.iter()) + .chain(date_tally.mismatches.iter()) + .chain(uint_tally.mismatches.iter()) .collect(); if !all.is_empty() { println!("\nMISMATCHES:"); @@ -501,6 +776,6 @@ fn main() -> ExitCode { } return ExitCode::FAILURE; } - println!("\nAll vectors validated against dcbor (Rust) {}", "0.25.2"); + println!("\nAll vectors validated against dcbor (Rust) 0.25.2 ({BUILD} build)"); ExitCode::SUCCESS } diff --git a/tests/tags-store.test.ts b/tests/tags-store.test.ts index e8f54fe..3e58c91 100644 --- a/tests/tags-store.test.ts +++ b/tests/tags-store.test.ts @@ -132,7 +132,123 @@ describe("TagsStore", () => { }); }); -describe("registerStandardTags: the bignum tags are opt-in (review N3)", () => { +describe("registerStandardTags registers unconditionally, like insert_all", () => { + it("moves the standard name back to the standard value", async () => { + const { registerStandardTags } = await import("../src/tags"); + const store = new TagsStore(); + store.register(Tag.from(1, "date")); + store.register(Tag.from(99, "date")); + expect(store.tagForName("date")?.value).toBe(99); + registerStandardTags(store); + expect(store.tagForName("date")?.value).toBe(1); + expect(store.nameForValue(99)).toBe("date"); // the by-value entry stays, as in the reference + expect(store.nameForValue(1)).toBe("date"); + }); + it("does the same for the bignum tags with { bignum: true }", async () => { + const { registerStandardTags } = await import("../src/tags"); + const store = new TagsStore(); + store.register(Tag.from(2, "positive-bignum")); + store.register(Tag.from(98, "positive-bignum")); + registerStandardTags(store, { bignum: true }); + expect(store.tagForName("positive-bignum")?.value).toBe(2); + expect(store.tagForName("negative-bignum")?.value).toBe(3); + }); + it("throws Custom for a conflicting name on tag 1 and sets no summarizer", async () => { + const { registerStandardTags } = await import("../src/tags"); + const { CborError } = await import("../src"); + const store = new TagsStore(); + store.register(Tag.from(1, "other")); + let error: unknown; + try { + registerStandardTags(store); + } catch (e) { + error = e; + } + expect(CborError.isCborError(error) && error.code).toBe("Custom"); + expect(CborError.isCborError(error) && error.message).toBe( + "Attempt to register tag: 1 'other' with different name: 'date'", + ); + expect(store.summarizer(1)).toBeUndefined(); + expect(store.nameForValue(1)).toBe("other"); + }); + it("registerAll accepts a readonly array or any iterable", () => { + const store = new TagsStore(); + const frozen: readonly Tag[] = Object.freeze([Tag.from(5, "five")]); + store.registerAll(frozen); + store.registerAll(new Set([Tag.from(6, "six")])); + expect(store.nameForValue(5)).toBe("five"); + expect(store.nameForValue(6)).toBe("six"); + }); +}); + +describe("tags are frozen values; the store keeps them by identity", () => { + it("Tag.from returns a frozen object", () => { + expect(Object.isFrozen(Tag.from(1, "date"))).toBe(true); + expect(Object.isFrozen(Tag.from(12345))).toBe(true); + }); + it("a mutable literal is copied, so later mutation does not reach the store", () => { + const store = new TagsStore(); + const literal = { value: 7, name: "seven" }; + store.register(literal); + literal.name = "changed"; + expect(store.nameForValue(7)).toBe("seven"); + expect(store.tagForName("seven")?.value).toBe(7); + expect(Object.isFrozen(store.tagForValue(7))).toBe(true); + }); + it("a frozen caller tag is stored by identity", async () => { + const { registerStandardTags } = await import("../src/tags"); + const store = new TagsStore(); + const eight = Tag.from(8, "eight"); + store.register(eight); + expect(store.tagForValue(8)).toBe(eight); + expect(store.tagForName("eight")).toBe(eight); + registerStandardTags(store); + expect(Object.isFrozen(store.tagForValue(1))).toBe(true); + }); +}); + +describe("TagsStore.clone mirrors #[derive(Clone)]", () => { + it("answers every lookup like the original and shares frozen tags and summarizers", () => { + const store = new TagsStore(); + const answer = Tag.from(42, "answer"); + store.register(answer); + const summarizer = () => ({ ok: true as const, value: "summary" }); + store.setSummarizer(42, summarizer); + const copy = store.clone(); + expect(copy).toBeInstanceOf(TagsStore); + expect(copy).not.toBe(store); + expect(copy.tagForValue(42)).toBe(answer); + expect(copy.tagForName("answer")).toBe(answer); + expect(copy.nameForValue(42)).toBe("answer"); + expect(copy.summarizer(42)).toBe(summarizer); + }); + it("registrations and summarizers set on one store do not reach the other", () => { + const store = new TagsStore(); + store.register(Tag.from(1, "one")); + const copy = store.clone(); + copy.register(Tag.from(2, "two")); + copy.setSummarizer(2, () => ({ ok: true as const, value: "two" })); + store.register(Tag.from(3, "three")); + expect(store.tagForValue(2)).toBeUndefined(); + expect(store.summarizer(2)).toBeUndefined(); + expect(copy.tagForValue(3)).toBeUndefined(); + expect(copy.nameForValue(1)).toBe("one"); + }); +}); + +describe("one global tags store per process", () => { + it("lives on globalThis under the registered symbol", async () => { + const { getGlobalTagsStore } = await import("../src"); + const slot = globalThis as { [k: symbol]: unknown }; + const key = Symbol.for("@blockchaincommons/dcbor/global-tags-store@1"); + const store = getGlobalTagsStore(); + expect(slot[key]).toBe(store); + expect(getGlobalTagsStore()).toBe(getGlobalTagsStore()); + expect(getGlobalTagsStore().clone()).not.toBe(getGlobalTagsStore()); + }); +}); + +describe("registerStandardTags: the bignum tags are opt-in", () => { it("names only the date tag by default, as the reference without num-bigint", async () => { const { registerStandardTags, TAG_DATE, TAG_POSITIVE_BIGNUM, TAG_NEGATIVE_BIGNUM } = await import("../src/tags"); @@ -152,7 +268,7 @@ describe("registerStandardTags: the bignum tags are opt-in (review N3)", () => { expect(store.tagForValue(TAG_NEGATIVE_BIGNUM)?.name).toBe("negative-bignum"); expect(store.summarizer(BigInt(TAG_POSITIVE_BIGNUM))).toBeDefined(); }); - it("a registration conflict is the package's CborError (review N4)", async () => { + it("a registration conflict is the package's CborError", async () => { const { CborError } = await import("../src"); const store = new TagsStore(); store.register(Tag.from(100, "first-name")); diff --git a/tests/unicode.test.ts b/tests/unicode.test.ts new file mode 100644 index 0000000..3ad0e6c --- /dev/null +++ b/tests/unicode.test.ts @@ -0,0 +1,23 @@ +/** + * Guard: the engine's Unicode tables must not be older than the + * reference's. + * + * Decoding rejects non-NFC text with `String.prototype.normalize`; the + * reference uses the `unicode-normalization` crate (0.1.25, Unicode 17). + * The two agree exhaustively when the engine implements the same Unicode + * version or a newer one; an older engine would accept or reject different + * strings. `process.versions.unicode` is reported by Node and Bun. + */ + +describe("engine Unicode version matches the reference's normalization tables", () => { + it("is Unicode 17 or newer (unicode-normalization 0.1.25)", () => { + const reported = (globalThis as { process?: { versions?: { unicode?: string } } }).process + ?.versions?.unicode; + if (reported === undefined) return; // an engine that does not report it + const major = Number(reported.split(".")[0]); + expect( + major, + `NFC tables must match unicode-normalization 0.1.25 (Unicode 17); engine reports ${reported}`, + ).toBeGreaterThanOrEqual(17); + }); +}); diff --git a/tests/vectors/date-corpus.ts b/tests/vectors/date-corpus.ts new file mode 100644 index 0000000..92e3dff --- /dev/null +++ b/tests/vectors/date-corpus.ts @@ -0,0 +1,127 @@ +/** + * Curated DATE corpus: `CborDate` decoding and display. + * + * - `decode` rows: tag-1 bytes -> `CborDate.fromTaggedCbor(decodeCbor(hex))` + * -> the re-encoded bytes and `toString()`, or the `CborError` code and + * message. The Rust harness runs `Date::from_tagged_cbor` on the same + * bytes; where the reference would panic (`from_timestamp` on ±Infinity + * or whole seconds outside chrono's range) it probes chrono first and + * emulates the port's `InvalidDate`. + * - `display` rows: a `date` / `datestr` recipe -> bytes and `toString()`, + * or the code. The reference renders `Display`. + * + * Expectations are generated with the working tree + * (`scripts/generate-vectors.ts`) into `tests/vectors/date-vectors.json`. + */ + +import type { Recipe } from "./recipes"; + +export type DateCorpusEntry = + | { name: string; kind: "decode"; hex: string; note?: string } + | { name: string; kind: "display"; recipe: Recipe; note?: string }; + +const decode = (name: string, hex: string, note?: string): DateCorpusEntry => + note === undefined ? { name, kind: "decode", hex } : { name, kind: "decode", hex, note }; +const display = (name: string, recipe: Recipe, note?: string): DateCorpusEntry => + note === undefined ? { name, kind: "display", recipe } : { name, kind: "display", recipe, note }; +const date = (seconds: number | string): Recipe => ({ k: "date", seconds: String(seconds) }); +const datestr = (v: string): Recipe => ({ k: "datestr", v }); + +export const dateCorpus: DateCorpusEntry[] = [ + // ---- decode: accepted instants + decode("decode/epoch", "c100"), + decode("decode/minus-one", "c120"), + decode("decode/2022-12-22", "c11a63a3b4c0"), + decode("decode/fraction-half", "c1fb41d962825c200000", "1703545200.5: a plain instant"), + decode("decode/f16-1.5", "c1f93e00"), + decode( + "decode/one-plus-epsilon", + "c1fb3ff0000000000001", + "sub-nanosecond fraction is dropped: re-encodes as 1", + ), + decode("decode/nan-saturates-to-epoch", "c1f97e00"), + decode("decode/max", "c11b000007779a0a6b7f"), + decode("decode/min", "c13b000007948cf211ff"), + decode( + "decode/min-minus-half-truncates", + "c1fbc29e5233c8480200", + "MIN - 0.5: the truncated whole seconds are MIN", + ), + decode("decode/max-plus-fraction", "c1fb429dde6829adffff"), + // ---- decode: rejected + decode("decode/infinity-throws", "c1f97c00", "the reference panics; emulated InvalidDate"), + decode( + "decode/negative-infinity-throws", + "c1f9fc00", + "the reference panics; emulated InvalidDate", + ), + decode( + "decode/2^60-beyond-range-throws", + "c11b1000000000000000", + "exact in f64, beyond chrono's range: the reference panics", + ), + decode("decode/max-plus-one-throws", "c11b000007779a0a6b80"), + decode("decode/min-minus-one-throws", "c13b000007948cf21200"), + decode( + "decode/i64-max-inexact-out-of-range", + "c11b7fffffffffffffff", + "f64::exact_from_u64 fails: OutOfRange", + ), + decode("decode/negative-inexact-out-of-range", "c13b7fffffffffffffff"), + decode("decode/65-bit-negative-out-of-range", "c13b80000000000007ff"), + decode("decode/text-content-wrong-type", "c16161"), + decode("decode/false-content-wrong-type", "c1f4"), + decode("decode/tag-of-tag-wrong-type", "c1c100"), + decode("decode/untagged-wrong-type", "6161"), + decode("decode/wrong-tag", "c26161"), + decode("decode/wrong-tag-40000", "d99c4000"), + decode("decode/truncated-underrun", "c1"), + decode("decode/non-canonical-numeric", "c1f94200"), + // ---- display: from_timestamp + display("display/midnight-with-fraction", date(1675814400.5)), + display("display/1.5", date(1.5)), + display("display/-0.5-truncates-toward-zero", date(-0.5)), + display("display/-1", date(-1)), + display("display/-86400", date(-86400)), + display("display/year--4-leap-day", date(-62288352000)), + display("display/year-0", date(-62167219200)), + display("display/year-50", date(-60589296000)), + display("display/9999-12-31", date(253402300799)), + display("display/year-10000", date(253402300800)), + display("display/year-12023", date(317245334400)), + display("display/year-100000", date(3093527980800)), + display("display/max", date(8210266876799)), + display("display/max-plus-0.999", date(8210266876799.999)), + display("display/min", date(-8334601228800)), + display("display/min-minus-0.5", date(-8334601228800.5)), + display("display/min-minus-0.999", date(-8334601228800.999)), + display("display/nan", date("NaN")), + display("display/infinity-throws", date("Infinity")), + display("display/max-plus-one-throws", date(8210266876800)), + display("display/min-minus-one-throws", date(-8334601228801)), + display("display/nanosecond-fraction-rounds-in-f64", date("1703500245.999999999")), + // ---- display: from_string + display("display/bare-date", datestr("2023-02-08")), + display("display/rfc3339", datestr("2023-02-08T15:30:45Z")), + display("display/rfc3339-offset", datestr("2023-02-08T15:30:45+05:30")), + display( + "display/rfc3339-fraction", + datestr("2023-12-25T10:30:45.999999999Z"), + "second 45 on display, 46 on the wire", + ), + display("display/leap-second", datestr("2023-12-25T10:30:60Z")), + display("display/leap-second-fraction-offset", datestr("2023-12-25T23:59:60.5+01:00")), + display("display/leap-second-end-of-year", datestr("2023-12-31T23:59:60Z")), + display("display/negative-year", datestr("-0001-01-01")), + display("display/five-digit-year", datestr("+12023-02-08")), + display("display/invalid-string-throws", datestr("not-a-date")), + display("display/second-61-throws", datestr("2023-12-25T10:30:61Z")), +]; + +{ + const seen = new Set(); + for (const { name } of dateCorpus) { + if (seen.has(name)) throw new Error(`duplicate date-corpus name: ${name}`); + seen.add(name); + } +} diff --git a/tests/vectors/date-vectors.json b/tests/vectors/date-vectors.json new file mode 100644 index 0000000..40dec76 --- /dev/null +++ b/tests/vectors/date-vectors.json @@ -0,0 +1,709 @@ +{ + "//": "GENERATED by scripts/generate-vectors.ts - do not edit by hand. Review regenerated expectations before committing them.", + "sourceCommit": "02766ae64e450096afd888380a8f80bcf24c2511", + "count": 60, + "vectors": [ + { + "name": "decode/epoch", + "kind": "decode", + "hex": "c100", + "expect": { + "ok": true, + "hex": "c100", + "display": "1970-01-01" + } + }, + { + "name": "decode/minus-one", + "kind": "decode", + "hex": "c120", + "expect": { + "ok": true, + "hex": "c120", + "display": "1969-12-31T23:59:59Z" + } + }, + { + "name": "decode/2022-12-22", + "kind": "decode", + "hex": "c11a63a3b4c0", + "expect": { + "ok": true, + "hex": "c11a63a3b4c0", + "display": "2022-12-22T01:37:04Z" + } + }, + { + "name": "decode/fraction-half", + "kind": "decode", + "hex": "c1fb41d962825c200000", + "expect": { + "ok": true, + "hex": "c1fb41d962825c200000", + "display": "2023-12-25T23:00:00Z" + }, + "note": "1703545200.5: a plain instant" + }, + { + "name": "decode/f16-1.5", + "kind": "decode", + "hex": "c1f93e00", + "expect": { + "ok": true, + "hex": "c1f93e00", + "display": "1970-01-01T00:00:01Z" + } + }, + { + "name": "decode/one-plus-epsilon", + "kind": "decode", + "hex": "c1fb3ff0000000000001", + "expect": { + "ok": true, + "hex": "c101", + "display": "1970-01-01T00:00:01Z" + }, + "note": "sub-nanosecond fraction is dropped: re-encodes as 1" + }, + { + "name": "decode/nan-saturates-to-epoch", + "kind": "decode", + "hex": "c1f97e00", + "expect": { + "ok": true, + "hex": "c100", + "display": "1970-01-01" + } + }, + { + "name": "decode/max", + "kind": "decode", + "hex": "c11b000007779a0a6b7f", + "expect": { + "ok": true, + "hex": "c11b000007779a0a6b7f", + "display": "+262142-12-31T23:59:59Z" + } + }, + { + "name": "decode/min", + "kind": "decode", + "hex": "c13b000007948cf211ff", + "expect": { + "ok": true, + "hex": "c13b000007948cf211ff", + "display": "-262143-01-01" + } + }, + { + "name": "decode/min-minus-half-truncates", + "kind": "decode", + "hex": "c1fbc29e5233c8480200", + "expect": { + "ok": true, + "hex": "c13b000007948cf211ff", + "display": "-262143-01-01" + }, + "note": "MIN - 0.5: the truncated whole seconds are MIN" + }, + { + "name": "decode/max-plus-fraction", + "kind": "decode", + "hex": "c1fb429dde6829adffff", + "expect": { + "ok": true, + "hex": "c1fb429dde6829adffff", + "display": "+262142-12-31T23:59:59Z" + } + }, + { + "name": "decode/infinity-throws", + "kind": "decode", + "hex": "c1f97c00", + "expect": { + "ok": false, + "code": "InvalidDate", + "message": "invalid ISO 8601 date string: non-finite timestamp" + }, + "note": "the reference panics; emulated InvalidDate" + }, + { + "name": "decode/negative-infinity-throws", + "kind": "decode", + "hex": "c1f9fc00", + "expect": { + "ok": false, + "code": "InvalidDate", + "message": "invalid ISO 8601 date string: non-finite timestamp" + }, + "note": "the reference panics; emulated InvalidDate" + }, + { + "name": "decode/2^60-beyond-range-throws", + "kind": "decode", + "hex": "c11b1000000000000000", + "expect": { + "ok": false, + "code": "InvalidDate", + "message": "invalid ISO 8601 date string: timestamp outside the representable range" + }, + "note": "exact in f64, beyond chrono's range: the reference panics" + }, + { + "name": "decode/max-plus-one-throws", + "kind": "decode", + "hex": "c11b000007779a0a6b80", + "expect": { + "ok": false, + "code": "InvalidDate", + "message": "invalid ISO 8601 date string: timestamp outside the representable range" + } + }, + { + "name": "decode/min-minus-one-throws", + "kind": "decode", + "hex": "c13b000007948cf21200", + "expect": { + "ok": false, + "code": "InvalidDate", + "message": "invalid ISO 8601 date string: timestamp outside the representable range" + } + }, + { + "name": "decode/i64-max-inexact-out-of-range", + "kind": "decode", + "hex": "c11b7fffffffffffffff", + "expect": { + "ok": false, + "code": "OutOfRange", + "message": "the CBOR numeric value could not be represented in the specified numeric type" + }, + "note": "f64::exact_from_u64 fails: OutOfRange" + }, + { + "name": "decode/negative-inexact-out-of-range", + "kind": "decode", + "hex": "c13b7fffffffffffffff", + "expect": { + "ok": false, + "code": "OutOfRange", + "message": "the CBOR numeric value could not be represented in the specified numeric type" + } + }, + { + "name": "decode/65-bit-negative-out-of-range", + "kind": "decode", + "hex": "c13b80000000000007ff", + "expect": { + "ok": false, + "code": "OutOfRange", + "message": "the CBOR numeric value could not be represented in the specified numeric type" + } + }, + { + "name": "decode/text-content-wrong-type", + "kind": "decode", + "hex": "c16161", + "expect": { + "ok": false, + "code": "WrongType", + "message": "the decoded CBOR value was not the expected type" + } + }, + { + "name": "decode/false-content-wrong-type", + "kind": "decode", + "hex": "c1f4", + "expect": { + "ok": false, + "code": "WrongType", + "message": "the decoded CBOR value was not the expected type" + } + }, + { + "name": "decode/tag-of-tag-wrong-type", + "kind": "decode", + "hex": "c1c100", + "expect": { + "ok": false, + "code": "WrongType", + "message": "the decoded CBOR value was not the expected type" + } + }, + { + "name": "decode/untagged-wrong-type", + "kind": "decode", + "hex": "6161", + "expect": { + "ok": false, + "code": "WrongType", + "message": "the decoded CBOR value was not the expected type" + } + }, + { + "name": "decode/wrong-tag", + "kind": "decode", + "hex": "c26161", + "expect": { + "ok": false, + "code": "WrongTag", + "message": "expected CBOR tag 1, but got 2" + } + }, + { + "name": "decode/wrong-tag-40000", + "kind": "decode", + "hex": "d99c4000", + "expect": { + "ok": false, + "code": "WrongTag", + "message": "expected CBOR tag 1, but got 40000" + } + }, + { + "name": "decode/truncated-underrun", + "kind": "decode", + "hex": "c1", + "expect": { + "ok": false, + "code": "Underrun", + "message": "early end of CBOR data" + } + }, + { + "name": "decode/non-canonical-numeric", + "kind": "decode", + "hex": "c1f94200", + "expect": { + "ok": false, + "code": "NonCanonicalNumeric", + "message": "a CBOR numeric value was encoded in non-canonical form" + } + }, + { + "name": "display/midnight-with-fraction", + "kind": "display", + "recipe": { + "k": "date", + "seconds": "1675814400.5" + }, + "expect": { + "ok": true, + "hex": "c1fb41d8f8b980200000", + "display": "2023-02-08" + } + }, + { + "name": "display/1.5", + "kind": "display", + "recipe": { + "k": "date", + "seconds": "1.5" + }, + "expect": { + "ok": true, + "hex": "c1f93e00", + "display": "1970-01-01T00:00:01Z" + } + }, + { + "name": "display/-0.5-truncates-toward-zero", + "kind": "display", + "recipe": { + "k": "date", + "seconds": "-0.5" + }, + "expect": { + "ok": true, + "hex": "c100", + "display": "1970-01-01" + } + }, + { + "name": "display/-1", + "kind": "display", + "recipe": { + "k": "date", + "seconds": "-1" + }, + "expect": { + "ok": true, + "hex": "c120", + "display": "1969-12-31T23:59:59Z" + } + }, + { + "name": "display/-86400", + "kind": "display", + "recipe": { + "k": "date", + "seconds": "-86400" + }, + "expect": { + "ok": true, + "hex": "c13a0001517f", + "display": "1969-12-31" + } + }, + { + "name": "display/year--4-leap-day", + "kind": "display", + "recipe": { + "k": "date", + "seconds": "-62288352000" + }, + "expect": { + "ok": true, + "hex": "c13b0000000e80acd2ff", + "display": "-0004-02-29" + } + }, + { + "name": "display/year-0", + "kind": "display", + "recipe": { + "k": "date", + "seconds": "-62167219200" + }, + "expect": { + "ok": true, + "hex": "c13b0000000e79747bff", + "display": "0000-01-01" + } + }, + { + "name": "display/year-50", + "kind": "display", + "recipe": { + "k": "date", + "seconds": "-60589296000" + }, + "expect": { + "ok": true, + "hex": "c13b0000000e1b67497f", + "display": "0050-01-01" + } + }, + { + "name": "display/9999-12-31", + "kind": "display", + "recipe": { + "k": "date", + "seconds": "253402300799" + }, + "expect": { + "ok": true, + "hex": "c11b0000003afff4417f", + "display": "9999-12-31T23:59:59Z" + } + }, + { + "name": "display/year-10000", + "kind": "display", + "recipe": { + "k": "date", + "seconds": "253402300800" + }, + "expect": { + "ok": true, + "hex": "c11b0000003afff44180", + "display": "+10000-01-01" + } + }, + { + "name": "display/year-12023", + "kind": "display", + "recipe": { + "k": "date", + "seconds": "317245334400" + }, + "expect": { + "ok": true, + "hex": "c11b00000049dd4ba380", + "display": "+12023-02-08" + } + }, + { + "name": "display/year-100000", + "kind": "display", + "recipe": { + "k": "date", + "seconds": "3093527980800" + }, + "expect": { + "ok": true, + "hex": "c11b000002d044a2eb00", + "display": "+100000-01-01" + } + }, + { + "name": "display/max", + "kind": "display", + "recipe": { + "k": "date", + "seconds": "8210266876799" + }, + "expect": { + "ok": true, + "hex": "c11b000007779a0a6b7f", + "display": "+262142-12-31T23:59:59Z" + } + }, + { + "name": "display/max-plus-0.999", + "kind": "display", + "recipe": { + "k": "date", + "seconds": "8210266876799.999" + }, + "expect": { + "ok": true, + "hex": "c1fb429dde6829adffff", + "display": "+262142-12-31T23:59:59Z" + } + }, + { + "name": "display/min", + "kind": "display", + "recipe": { + "k": "date", + "seconds": "-8334601228800" + }, + "expect": { + "ok": true, + "hex": "c13b000007948cf211ff", + "display": "-262143-01-01" + } + }, + { + "name": "display/min-minus-0.5", + "kind": "display", + "recipe": { + "k": "date", + "seconds": "-8334601228800.5" + }, + "expect": { + "ok": true, + "hex": "c13b000007948cf211ff", + "display": "-262143-01-01" + } + }, + { + "name": "display/min-minus-0.999", + "kind": "display", + "recipe": { + "k": "date", + "seconds": "-8334601228800.999" + }, + "expect": { + "ok": true, + "hex": "c13b000007948cf211ff", + "display": "-262143-01-01" + } + }, + { + "name": "display/nan", + "kind": "display", + "recipe": { + "k": "date", + "seconds": "NaN" + }, + "expect": { + "ok": true, + "hex": "c100", + "display": "1970-01-01" + } + }, + { + "name": "display/infinity-throws", + "kind": "display", + "recipe": { + "k": "date", + "seconds": "Infinity" + }, + "expect": { + "ok": false, + "code": "InvalidDate" + } + }, + { + "name": "display/max-plus-one-throws", + "kind": "display", + "recipe": { + "k": "date", + "seconds": "8210266876800" + }, + "expect": { + "ok": false, + "code": "InvalidDate" + } + }, + { + "name": "display/min-minus-one-throws", + "kind": "display", + "recipe": { + "k": "date", + "seconds": "-8334601228801" + }, + "expect": { + "ok": false, + "code": "InvalidDate" + } + }, + { + "name": "display/nanosecond-fraction-rounds-in-f64", + "kind": "display", + "recipe": { + "k": "date", + "seconds": "1703500245.999999999" + }, + "expect": { + "ok": true, + "hex": "c11a658959d6", + "display": "2023-12-25T10:30:46Z" + } + }, + { + "name": "display/bare-date", + "kind": "display", + "recipe": { + "k": "datestr", + "v": "2023-02-08" + }, + "expect": { + "ok": true, + "hex": "c11a63e2e600", + "display": "2023-02-08" + } + }, + { + "name": "display/rfc3339", + "kind": "display", + "recipe": { + "k": "datestr", + "v": "2023-02-08T15:30:45Z" + }, + "expect": { + "ok": true, + "hex": "c11a63e3c025", + "display": "2023-02-08T15:30:45Z" + } + }, + { + "name": "display/rfc3339-offset", + "kind": "display", + "recipe": { + "k": "datestr", + "v": "2023-02-08T15:30:45+05:30" + }, + "expect": { + "ok": true, + "hex": "c11a63e372cd", + "display": "2023-02-08T10:00:45Z" + } + }, + { + "name": "display/rfc3339-fraction", + "kind": "display", + "recipe": { + "k": "datestr", + "v": "2023-12-25T10:30:45.999999999Z" + }, + "expect": { + "ok": true, + "hex": "c11a658959d6", + "display": "2023-12-25T10:30:45Z" + }, + "note": "second 45 on display, 46 on the wire" + }, + { + "name": "display/leap-second", + "kind": "display", + "recipe": { + "k": "datestr", + "v": "2023-12-25T10:30:60Z" + }, + "expect": { + "ok": true, + "hex": "c11a658959e4", + "display": "2023-12-25T10:30:60Z" + } + }, + { + "name": "display/leap-second-fraction-offset", + "kind": "display", + "recipe": { + "k": "datestr", + "v": "2023-12-25T23:59:60.5+01:00" + }, + "expect": { + "ok": true, + "hex": "c1fb41d962825c200000", + "display": "2023-12-25T22:59:60Z" + } + }, + { + "name": "display/leap-second-end-of-year", + "kind": "display", + "recipe": { + "k": "datestr", + "v": "2023-12-31T23:59:60Z" + }, + "expect": { + "ok": true, + "hex": "c11a65920080", + "display": "2023-12-31T23:59:60Z" + } + }, + { + "name": "display/negative-year", + "kind": "display", + "recipe": { + "k": "datestr", + "v": "-0001-01-01" + }, + "expect": { + "ok": true, + "hex": "c13b0000000e7b55af7f", + "display": "-0001-01-01" + } + }, + { + "name": "display/five-digit-year", + "kind": "display", + "recipe": { + "k": "datestr", + "v": "+12023-02-08" + }, + "expect": { + "ok": true, + "hex": "c11b00000049dd4ba380", + "display": "+12023-02-08" + } + }, + { + "name": "display/invalid-string-throws", + "kind": "display", + "recipe": { + "k": "datestr", + "v": "not-a-date" + }, + "expect": { + "ok": false, + "code": "InvalidDate" + } + }, + { + "name": "display/second-61-throws", + "kind": "display", + "recipe": { + "k": "datestr", + "v": "2023-12-25T10:30:61Z" + }, + "expect": { + "ok": false, + "code": "InvalidDate" + } + } + ] +} diff --git a/tests/vectors/decode-corpus.ts b/tests/vectors/decode-corpus.ts index 9355892..32c8532 100644 --- a/tests/vectors/decode-corpus.ts +++ b/tests/vectors/decode-corpus.ts @@ -1,16 +1,16 @@ /** - * Curated golden DECODE corpus (API_REDESIGN_PLAN P1.1a). + * Curated golden DECODE corpus. * * Byte sequences with the outcome `decodeCbor` must produce: acceptance - * (in which case decode->re-encode must reproduce the input bytes exactly) - * or rejection with a specific `CborError.code`. Covers every reachable - * throw site in src/decode.ts (including the `checkCanonicalEncoding` - * re-encode rejections and CborMap.setNext map-ordering errors), each - * error propagated through nested containers, and the canonical-form - * accepts that are easy to get wrong. + * (in which case decode->re-encode must reproduce the input bytes exactly, + * unless the entry pins a different `expect.hex` - whole-valued f32/f64 heads + * decode to integer nodes, as in the reference) or rejection with a specific + * `CborError.code`. Covers every reachable throw site in src/decode.ts + * (including the `checkCanonicalEncoding` re-encode rejections and + * CborMap.setNext map-ordering errors), each error propagated through nested + * containers, and the canonical-form accepts that are easy to get wrong. * - * All vectors were machine-verified against the pre-redesign build; the - * generator (`scripts/generate-vectors.mjs`) re-verifies each expectation + * The generator (`scripts/generate-vectors.ts`) re-verifies each expectation * and fails loudly on any mismatch before writing * `tests/vectors/decode-vectors.json`. */ @@ -18,7 +18,11 @@ export interface DecodeCorpusEntry { name: string; hex: string; - expect: { ok: true } | { ok: false; code: string }; + /** + * `ok: true` accepts and re-encodes to `hex` unless `expect.hex` names the + * bytes the decoded node re-encodes to instead. + */ + expect: { ok: true; hex?: string } | { ok: false; code: string }; note: string; } @@ -27,19 +31,19 @@ export const decodeCorpus: DecodeCorpusEntry[] = [ name: "reject/Underrun/(empty)", hex: "", expect: { ok: false, code: "Underrun" }, - note: "empty input: readCbor L159 remaining<1", + note: "empty input", }, { name: "reject/Underrun/18", hex: "18", expect: { ok: false, code: "Underrun" }, - note: "uint 2-byte head, 0 of 1 arg bytes (L103)", + note: "uint 2-byte head, 0 of 1 arg bytes", }, { name: "reject/Underrun/19", hex: "19", expect: { ok: false, code: "Underrun" }, - note: "uint 3-byte head, 0 of 2 arg bytes (L112)", + note: "uint 3-byte head, 0 of 2 arg bytes", }, { name: "reject/Underrun/1900", @@ -51,13 +55,13 @@ export const decodeCorpus: DecodeCorpusEntry[] = [ name: "reject/Underrun/1a000000", hex: "1a000000", expect: { ok: false, code: "Underrun" }, - note: "uint 5-byte head, 3 of 4 arg bytes (L122)", + note: "uint 5-byte head, 3 of 4 arg bytes", }, { name: "reject/Underrun/1b00000000000000", hex: "1b00000000000000", expect: { ok: false, code: "Underrun" }, - note: "uint 9-byte head, 7 of 8 arg bytes (L134)", + note: "uint 9-byte head, 7 of 8 arg bytes", }, { name: "reject/Underrun/38", @@ -111,7 +115,7 @@ export const decodeCorpus: DecodeCorpusEntry[] = [ name: "reject/Underrun/f9", hex: "f9", expect: { ok: false, code: "Underrun" }, - note: "f16 head with no payload (hv25 dataRemaining<2)", + note: "f16 head with no payload", }, { name: "reject/Underrun/f97e", @@ -135,7 +139,7 @@ export const decodeCorpus: DecodeCorpusEntry[] = [ name: "reject/Underrun/41", hex: "41", expect: { ok: false, code: "Underrun" }, - note: "bytestring declares 1 body byte, has 0 (L181)", + note: "bytestring declares 1 body byte, has 0", }, { name: "reject/Underrun/4401ff02", @@ -147,7 +151,7 @@ export const decodeCorpus: DecodeCorpusEntry[] = [ name: "reject/Underrun/61", hex: "61", expect: { ok: false, code: "Underrun" }, - note: "text declares 1 body byte, has 0 (L192)", + note: "text declares 1 body byte, has 0", }, { name: "reject/Underrun/626f", @@ -213,7 +217,7 @@ export const decodeCorpus: DecodeCorpusEntry[] = [ name: "reject/UnsupportedHeaderValue/1c", hex: "1c", expect: { ok: false, code: "UnsupportedHeaderValue" }, - note: "major 0 headerValue 28 (L152), details.headerValue=28", + note: "major 0 headerValue 28, details.headerValue=28", }, { name: "reject/UnsupportedHeaderValue/1d", @@ -393,7 +397,7 @@ export const decodeCorpus: DecodeCorpusEntry[] = [ name: "reject/NonCanonicalNumeric/1800", hex: "1800", expect: { ok: false, code: "NonCanonicalNumeric" }, - note: "uint 0 in 2-byte head (value<24, L107)", + note: "uint 0 in 2-byte head (value<24)", }, { name: "reject/NonCanonicalNumeric/1817", @@ -405,7 +409,7 @@ export const decodeCorpus: DecodeCorpusEntry[] = [ name: "reject/NonCanonicalNumeric/190017", hex: "190017", expect: { ok: false, code: "NonCanonicalNumeric" }, - note: "uint 23 in 3-byte head (value<=0xff, L117)", + note: "uint 23 in 3-byte head (value<=0xff)", }, { name: "reject/NonCanonicalNumeric/1900ff", @@ -417,7 +421,7 @@ export const decodeCorpus: DecodeCorpusEntry[] = [ name: "reject/NonCanonicalNumeric/1a00000017", hex: "1a00000017", expect: { ok: false, code: "NonCanonicalNumeric" }, - note: "uint 23 in 5-byte head (value<=0xffff, L129)", + note: "uint 23 in 5-byte head (value<=0xffff)", }, { name: "reject/NonCanonicalNumeric/1a0000ffff", @@ -429,7 +433,7 @@ export const decodeCorpus: DecodeCorpusEntry[] = [ name: "reject/NonCanonicalNumeric/1b0000000000000017", hex: "1b0000000000000017", expect: { ok: false, code: "NonCanonicalNumeric" }, - note: "uint 23 in 9-byte head (value<=0xffffffff, L147)", + note: "uint 23 in 9-byte head (value<=0xffffffff)", }, { name: "reject/NonCanonicalNumeric/1b00000000ffffffff", @@ -543,7 +547,7 @@ export const decodeCorpus: DecodeCorpusEntry[] = [ name: "reject/NonCanonicalNumeric/fb4045000000000000", hex: "fb4045000000000000", expect: { ok: false, code: "NonCanonicalNumeric" }, - note: "42.0 as f64: re-encode reduces to int 182a (checkCanonicalEncoding L306)", + note: "42.0 as f64: re-encode reduces to int 182a (checkCanonicalEncoding)", }, { name: "reject/NonCanonicalNumeric/fa42280000", @@ -599,6 +603,24 @@ export const decodeCorpus: DecodeCorpusEntry[] = [ expect: { ok: false, code: "NonCanonicalNumeric" }, note: "2^63 as f64 reduces to uint 1b8000000000000000 (bigint reduction path)", }, + { + name: "reject/NonCanonicalNumeric/fa4f000000", + hex: "fa4f000000", + expect: { ok: false, code: "NonCanonicalNumeric" }, + note: "2^31 as f32: `as i32` saturates to i32::MAX, which rounds back to exactly 2^31 -> whole -> rejected", + }, + { + name: "reject/NonCanonicalNumeric/facf000000", + hex: "facf000000", + expect: { ok: false, code: "NonCanonicalNumeric" }, + note: "-2^31 as f32: i32::MIN round-trips exactly -> rejected", + }, + { + name: "reject/NonCanonicalNumeric/fbc3e0000000000000", + hex: "fbc3e0000000000000", + expect: { ok: false, code: "NonCanonicalNumeric" }, + note: "-2^63 as f64: i64::MIN round-trips exactly -> rejected", + }, { name: "reject/NonCanonicalNumeric/fa3fc00000", hex: "fa3fc00000", @@ -809,6 +831,57 @@ export const decodeCorpus: DecodeCorpusEntry[] = [ expect: { ok: false, code: "InvalidUtf8" }, note: "0xff is never valid in UTF-8", }, + // Utf8Error message coverage: a bad second byte, a bad later + // byte (error_len 2/3), an error after valid text (valid_up_to 3), and + // truncation inside a sequence ("incomplete"). + { + name: "reject/InvalidUtf8/63e28228", + hex: "63e28228", + expect: { ok: false, code: "InvalidUtf8" }, + note: "3-byte lead, valid second byte, invalid third: error_len 2", + }, + { + name: "reject/InvalidUtf8/64f09f9841", + hex: "64f09f9841", + expect: { ok: false, code: "InvalidUtf8" }, + note: "4-byte lead, two valid continuations, invalid fourth: error_len 3", + }, + { + name: "reject/InvalidUtf8/63e0a041", + hex: "63e0a041", + expect: { ok: false, code: "InvalidUtf8" }, + note: "e0 with valid second byte a0, invalid third: error_len 2", + }, + { + name: "reject/InvalidUtf8/62f0a0", + hex: "62f0a0", + expect: { ok: false, code: "InvalidUtf8" }, + note: "4-byte sequence cut after a valid second byte: incomplete", + }, + { + name: "reject/InvalidUtf8/6441c3a9ff", + hex: "6441c3a9ff", + expect: { ok: false, code: "InvalidUtf8" }, + note: "valid 'Aé' then ff: valid_up_to 3", + }, + { + name: "reject/InvalidUtf8/64616263ff", + hex: "64616263ff", + expect: { ok: false, code: "InvalidUtf8" }, + note: "valid 'abc' then ff: valid_up_to 3", + }, + { + name: "reject/InvalidUtf8/62eda0", + hex: "62eda0", + expect: { ok: false, code: "InvalidUtf8" }, + note: "ed with second byte a0 (surrogate range) and no third: the present bad byte beats incomplete", + }, + { + name: "reject/InvalidUtf8/62c0af", + hex: "62c0af", + expect: { ok: false, code: "InvalidUtf8" }, + note: "overlong lead c0: error_len 1 whatever follows", + }, { name: "reject/NonCanonicalString/6365cc81", hex: "6365cc81", @@ -1214,6 +1287,82 @@ export const decodeCorpus: DecodeCorpusEntry[] = [ expect: { ok: true }, note: "2^64 exactly as f32 - whole-valued but exceeds u64 by 1, so no integer reduction", }, + // Whole-valued f32/f64 heads beyond the saturating-cast bounds (Rust + // `validate_canonical_f32/f64` + `From`): accepted, and the + // node is the integer the value reduces to (or stays a float when no + // integer fits), so the re-encoding may differ from the input. + { + name: "accept/fa4f000001-as-1a80000100", + hex: "fa4f000001", + expect: { ok: true, hex: "1a80000100" }, + note: "2^31+256: `as i32` saturates to 2^31 != value, so accepted; From -> Unsigned(2147483904)", + }, + { + name: "accept/fa4f7fffff-as-1affffff00", + hex: "fa4f7fffff", + expect: { ok: true, hex: "1affffff00" }, + note: "2^32-256 (largest f32 below 2^32) -> Unsigned(4294967040)", + }, + { + name: "accept/facf000001-as-3a80000100", + hex: "facf000001", + expect: { ok: true, hex: "3a80000100" }, + note: "-(2^31+256): -1f32 - n rounds to 2^31+256 -> Negative(2147483904) = -2147483905", + }, + { + name: "accept/fadf000000-as-3b8000000000000000", + hex: "fadf000000", + expect: { ok: true, hex: "3b8000000000000000" }, + note: "-2^63 as f32: -1f32 - n rounds to 2^63 -> Negative(2^63) = -9223372036854775809 (65-bit)", + }, + { + name: "accept/fadb000000-as-3b0080000000000000", + hex: "fadb000000", + expect: { ok: true, hex: "3b0080000000000000" }, + note: "-2^55 as f32: -1f32 - n rounds to 2^55 -> Negative(2^55) = -36028797018963969", + }, + { + name: "accept/fb43e0000000000001-as-1b8000000000000800", + hex: "fb43e0000000000001", + expect: { ok: true, hex: "1b8000000000000800" }, + note: "2^63+2048: `as i64` saturates to i64::MAX -> 2^63 != value, accepted; From -> Unsigned", + }, + { + name: "accept/fbc3e0000000000001-as-3b80000000000007ff", + hex: "fbc3e0000000000001", + expect: { ok: true, hex: "3b80000000000007ff" }, + note: "-(2^63+2048): accepted via i64 saturation; From -> Negative(2^63+2047)", + }, + { + name: "accept/fa4f800000", + hex: "fa4f800000", + expect: { ok: true }, + note: "2^32 as f32: accepted (saturating i32 image differs); no u32 fits, so the node stays a float", + }, + { + name: "accept/fa4fc00000", + hex: "fa4fc00000", + expect: { ok: true }, + note: "1.5*2^32 as f32: whole, accepted, stays a float", + }, + { + name: "accept/fa5a000000", + hex: "fa5a000000", + expect: { ok: true }, + note: "2^53 as f32: whole, accepted, stays a float", + }, + { + name: "accept/fa5f000000", + hex: "fa5f000000", + expect: { ok: true }, + note: "2^63 as f32: whole, accepted, stays a float", + }, + { + name: "accept/fadf800000", + hex: "fadf800000", + expect: { ok: true }, + note: "-2^64 as f32: -1f32 - n rounds to 2^64, which no u64 holds, so the node stays a float", + }, { name: "accept/fb3ff199999999999a", hex: "fb3ff199999999999a", @@ -1269,6 +1418,24 @@ export const decodeCorpus: DecodeCorpusEntry[] = [ expect: { ok: true }, note: "NFC-composed é (U+00E9) - passes both UTF-8 and NFC checks", }, + { + name: "accept/64efbbbf61", + hex: "64efbbbf61", + expect: { ok: true }, + note: 'text "\uFEFFa": a leading BOM is a character, kept on decode (Rust String::from_utf8 parity)', + }, + { + name: "accept/63efbbbf", + hex: "63efbbbf", + expect: { ok: true }, + note: "text consisting of only U+FEFF - re-encodes to the same 3 bytes, not to the empty string", + }, + { + name: "accept/8263efbbbf63efbbbf", + hex: "8263efbbbf63efbbbf", + expect: { ok: true }, + note: "two BOM-only strings in an array - each element keeps its BOM", + }, { name: "accept/8181818100", hex: "8181818100", diff --git a/tests/vectors/decode-vectors.json b/tests/vectors/decode-vectors.json index d433fc2..8cfebb5 100644 --- a/tests/vectors/decode-vectors.json +++ b/tests/vectors/decode-vectors.json @@ -1,41 +1,45 @@ { - "//": "GENERATED by scripts/generate-vectors.ts - do not edit by hand. Regeneration is a deliberate, reviewed act (see API_REDESIGN_PLAN.md P1.1).", - "sourceCommit": "29f9ec7ec58dee71dfdcc1e4540b39ca75ee9194", - "count": 241, + "//": "GENERATED by scripts/generate-vectors.ts - do not edit by hand. Review regenerated expectations before committing them.", + "sourceCommit": "02766ae64e450096afd888380a8f80bcf24c2511", + "count": 267, "vectors": [ { "name": "reject/Underrun/(empty)", "hex": "", "expect": { "ok": false, - "code": "Underrun" + "code": "Underrun", + "message": "early end of CBOR data" }, - "note": "empty input: readCbor L159 remaining<1" + "note": "empty input" }, { "name": "reject/Underrun/18", "hex": "18", "expect": { "ok": false, - "code": "Underrun" + "code": "Underrun", + "message": "early end of CBOR data" }, - "note": "uint 2-byte head, 0 of 1 arg bytes (L103)" + "note": "uint 2-byte head, 0 of 1 arg bytes" }, { "name": "reject/Underrun/19", "hex": "19", "expect": { "ok": false, - "code": "Underrun" + "code": "Underrun", + "message": "early end of CBOR data" }, - "note": "uint 3-byte head, 0 of 2 arg bytes (L112)" + "note": "uint 3-byte head, 0 of 2 arg bytes" }, { "name": "reject/Underrun/1900", "hex": "1900", "expect": { "ok": false, - "code": "Underrun" + "code": "Underrun", + "message": "early end of CBOR data" }, "note": "uint 3-byte head, 1 of 2 arg bytes" }, @@ -44,25 +48,28 @@ "hex": "1a000000", "expect": { "ok": false, - "code": "Underrun" + "code": "Underrun", + "message": "early end of CBOR data" }, - "note": "uint 5-byte head, 3 of 4 arg bytes (L122)" + "note": "uint 5-byte head, 3 of 4 arg bytes" }, { "name": "reject/Underrun/1b00000000000000", "hex": "1b00000000000000", "expect": { "ok": false, - "code": "Underrun" + "code": "Underrun", + "message": "early end of CBOR data" }, - "note": "uint 9-byte head, 7 of 8 arg bytes (L134)" + "note": "uint 9-byte head, 7 of 8 arg bytes" }, { "name": "reject/Underrun/38", "hex": "38", "expect": { "ok": false, - "code": "Underrun" + "code": "Underrun", + "message": "early end of CBOR data" }, "note": "negative 2-byte head truncated" }, @@ -71,7 +78,8 @@ "hex": "3b", "expect": { "ok": false, - "code": "Underrun" + "code": "Underrun", + "message": "early end of CBOR data" }, "note": "negative 9-byte head truncated" }, @@ -80,7 +88,8 @@ "hex": "58", "expect": { "ok": false, - "code": "Underrun" + "code": "Underrun", + "message": "early end of CBOR data" }, "note": "bytestring length head truncated" }, @@ -89,7 +98,8 @@ "hex": "78", "expect": { "ok": false, - "code": "Underrun" + "code": "Underrun", + "message": "early end of CBOR data" }, "note": "text length head truncated" }, @@ -98,7 +108,8 @@ "hex": "98", "expect": { "ok": false, - "code": "Underrun" + "code": "Underrun", + "message": "early end of CBOR data" }, "note": "array count head truncated" }, @@ -107,7 +118,8 @@ "hex": "b8", "expect": { "ok": false, - "code": "Underrun" + "code": "Underrun", + "message": "early end of CBOR data" }, "note": "map count head truncated" }, @@ -116,7 +128,8 @@ "hex": "d8", "expect": { "ok": false, - "code": "Underrun" + "code": "Underrun", + "message": "early end of CBOR data" }, "note": "tag head truncated" }, @@ -125,7 +138,8 @@ "hex": "f8", "expect": { "ok": false, - "code": "Underrun" + "code": "Underrun", + "message": "early end of CBOR data" }, "note": "simple 2-byte head truncated" }, @@ -134,16 +148,18 @@ "hex": "f9", "expect": { "ok": false, - "code": "Underrun" + "code": "Underrun", + "message": "early end of CBOR data" }, - "note": "f16 head with no payload (hv25 dataRemaining<2)" + "note": "f16 head with no payload" }, { "name": "reject/Underrun/f97e", "hex": "f97e", "expect": { "ok": false, - "code": "Underrun" + "code": "Underrun", + "message": "early end of CBOR data" }, "note": "f16 payload 1 of 2 bytes" }, @@ -152,7 +168,8 @@ "hex": "fa3fc000", "expect": { "ok": false, - "code": "Underrun" + "code": "Underrun", + "message": "early end of CBOR data" }, "note": "f32 payload 3 of 4 bytes" }, @@ -161,7 +178,8 @@ "hex": "fb3ff19999999999", "expect": { "ok": false, - "code": "Underrun" + "code": "Underrun", + "message": "early end of CBOR data" }, "note": "f64 payload 7 of 8 bytes" }, @@ -170,16 +188,18 @@ "hex": "41", "expect": { "ok": false, - "code": "Underrun" + "code": "Underrun", + "message": "early end of CBOR data" }, - "note": "bytestring declares 1 body byte, has 0 (L181)" + "note": "bytestring declares 1 body byte, has 0" }, { "name": "reject/Underrun/4401ff02", "hex": "4401ff02", "expect": { "ok": false, - "code": "Underrun" + "code": "Underrun", + "message": "early end of CBOR data" }, "note": "bytestring declares 4 body bytes, has 3" }, @@ -188,16 +208,18 @@ "hex": "61", "expect": { "ok": false, - "code": "Underrun" + "code": "Underrun", + "message": "early end of CBOR data" }, - "note": "text declares 1 body byte, has 0 (L192)" + "note": "text declares 1 body byte, has 0" }, { "name": "reject/Underrun/626f", "hex": "626f", "expect": { "ok": false, - "code": "Underrun" + "code": "Underrun", + "message": "early end of CBOR data" }, "note": "text declares 2 body bytes, has 1" }, @@ -206,7 +228,8 @@ "hex": "81", "expect": { "ok": false, - "code": "Underrun" + "code": "Underrun", + "message": "early end of CBOR data" }, "note": "array of 1 with no child; nested readCbor hits empty" }, @@ -215,7 +238,8 @@ "hex": "8201", "expect": { "ok": false, - "code": "Underrun" + "code": "Underrun", + "message": "early end of CBOR data" }, "note": "array of 2 missing second child" }, @@ -224,7 +248,8 @@ "hex": "a1", "expect": { "ok": false, - "code": "Underrun" + "code": "Underrun", + "message": "early end of CBOR data" }, "note": "map of 1 missing key" }, @@ -233,7 +258,8 @@ "hex": "a101", "expect": { "ok": false, - "code": "Underrun" + "code": "Underrun", + "message": "early end of CBOR data" }, "note": "map of 1 missing value" }, @@ -242,7 +268,8 @@ "hex": "c1", "expect": { "ok": false, - "code": "Underrun" + "code": "Underrun", + "message": "early end of CBOR data" }, "note": "tag 1 missing content item" }, @@ -251,7 +278,8 @@ "hex": "5b0000000100000000", "expect": { "ok": false, - "code": "Underrun" + "code": "Underrun", + "message": "early end of CBOR data" }, "note": "bytestring length 2^32 is a plain number (< 2^53); body missing -> Underrun" }, @@ -260,7 +288,8 @@ "hex": "9bffffffffffffffff", "expect": { "ok": false, - "code": "Underrun" + "code": "Underrun", + "message": "early end of CBOR data" }, "note": "array count 2^64-1 (bigint) loops fine, first missing child underruns" }, @@ -269,7 +298,8 @@ "hex": "bbffffffffffffffff", "expect": { "ok": false, - "code": "Underrun" + "code": "Underrun", + "message": "early end of CBOR data" }, "note": "map count bigint, first missing key underruns" }, @@ -278,7 +308,8 @@ "hex": "dbffffffffffffffff", "expect": { "ok": false, - "code": "Underrun" + "code": "Underrun", + "message": "early end of CBOR data" }, "note": "tag value 2^64-1 legal (bigint), missing content underruns" }, @@ -287,16 +318,18 @@ "hex": "1c", "expect": { "ok": false, - "code": "UnsupportedHeaderValue" + "code": "UnsupportedHeaderValue", + "message": "unsupported value in CBOR header" }, - "note": "major 0 headerValue 28 (L152), details.headerValue=28" + "note": "major 0 headerValue 28, details.headerValue=28" }, { "name": "reject/UnsupportedHeaderValue/1d", "hex": "1d", "expect": { "ok": false, - "code": "UnsupportedHeaderValue" + "code": "UnsupportedHeaderValue", + "message": "unsupported value in CBOR header" }, "note": "major 0 headerValue 29" }, @@ -305,7 +338,8 @@ "hex": "1e", "expect": { "ok": false, - "code": "UnsupportedHeaderValue" + "code": "UnsupportedHeaderValue", + "message": "unsupported value in CBOR header" }, "note": "major 0 headerValue 30" }, @@ -314,7 +348,8 @@ "hex": "1f", "expect": { "ok": false, - "code": "UnsupportedHeaderValue" + "code": "UnsupportedHeaderValue", + "message": "unsupported value in CBOR header" }, "note": "major 0 headerValue 31 (would-be indefinite)" }, @@ -323,7 +358,8 @@ "hex": "3c", "expect": { "ok": false, - "code": "UnsupportedHeaderValue" + "code": "UnsupportedHeaderValue", + "message": "unsupported value in CBOR header" }, "note": "negative headerValue 28" }, @@ -332,7 +368,8 @@ "hex": "3f", "expect": { "ok": false, - "code": "UnsupportedHeaderValue" + "code": "UnsupportedHeaderValue", + "message": "unsupported value in CBOR header" }, "note": "negative headerValue 31" }, @@ -341,7 +378,8 @@ "hex": "5f", "expect": { "ok": false, - "code": "UnsupportedHeaderValue" + "code": "UnsupportedHeaderValue", + "message": "unsupported value in CBOR header" }, "note": "indefinite-length bytestring" }, @@ -350,7 +388,8 @@ "hex": "7f", "expect": { "ok": false, - "code": "UnsupportedHeaderValue" + "code": "UnsupportedHeaderValue", + "message": "unsupported value in CBOR header" }, "note": "indefinite-length text" }, @@ -359,7 +398,8 @@ "hex": "9f", "expect": { "ok": false, - "code": "UnsupportedHeaderValue" + "code": "UnsupportedHeaderValue", + "message": "unsupported value in CBOR header" }, "note": "indefinite-length array" }, @@ -368,7 +408,8 @@ "hex": "bf", "expect": { "ok": false, - "code": "UnsupportedHeaderValue" + "code": "UnsupportedHeaderValue", + "message": "unsupported value in CBOR header" }, "note": "indefinite-length map" }, @@ -377,7 +418,8 @@ "hex": "df", "expect": { "ok": false, - "code": "UnsupportedHeaderValue" + "code": "UnsupportedHeaderValue", + "message": "unsupported value in CBOR header" }, "note": "tag headerValue 31" }, @@ -386,7 +428,8 @@ "hex": "fc", "expect": { "ok": false, - "code": "UnsupportedHeaderValue" + "code": "UnsupportedHeaderValue", + "message": "unsupported value in CBOR header" }, "note": "simple headerValue 28" }, @@ -395,7 +438,8 @@ "hex": "fd", "expect": { "ok": false, - "code": "UnsupportedHeaderValue" + "code": "UnsupportedHeaderValue", + "message": "unsupported value in CBOR header" }, "note": "simple headerValue 29" }, @@ -404,7 +448,8 @@ "hex": "fe", "expect": { "ok": false, - "code": "UnsupportedHeaderValue" + "code": "UnsupportedHeaderValue", + "message": "unsupported value in CBOR header" }, "note": "simple headerValue 30" }, @@ -413,7 +458,8 @@ "hex": "5c", "expect": { "ok": false, - "code": "UnsupportedHeaderValue" + "code": "UnsupportedHeaderValue", + "message": "unsupported value in CBOR header" }, "note": "bytestring headerValue 28" }, @@ -422,7 +468,8 @@ "hex": "5d", "expect": { "ok": false, - "code": "UnsupportedHeaderValue" + "code": "UnsupportedHeaderValue", + "message": "unsupported value in CBOR header" }, "note": "bytestring headerValue 29" }, @@ -431,7 +478,8 @@ "hex": "5e", "expect": { "ok": false, - "code": "UnsupportedHeaderValue" + "code": "UnsupportedHeaderValue", + "message": "unsupported value in CBOR header" }, "note": "bytestring headerValue 30" }, @@ -440,7 +488,8 @@ "hex": "7c", "expect": { "ok": false, - "code": "UnsupportedHeaderValue" + "code": "UnsupportedHeaderValue", + "message": "unsupported value in CBOR header" }, "note": "text headerValue 28" }, @@ -449,7 +498,8 @@ "hex": "7d", "expect": { "ok": false, - "code": "UnsupportedHeaderValue" + "code": "UnsupportedHeaderValue", + "message": "unsupported value in CBOR header" }, "note": "text headerValue 29" }, @@ -458,7 +508,8 @@ "hex": "7e", "expect": { "ok": false, - "code": "UnsupportedHeaderValue" + "code": "UnsupportedHeaderValue", + "message": "unsupported value in CBOR header" }, "note": "text headerValue 30" }, @@ -467,7 +518,8 @@ "hex": "9c", "expect": { "ok": false, - "code": "UnsupportedHeaderValue" + "code": "UnsupportedHeaderValue", + "message": "unsupported value in CBOR header" }, "note": "array headerValue 28" }, @@ -476,7 +528,8 @@ "hex": "9d", "expect": { "ok": false, - "code": "UnsupportedHeaderValue" + "code": "UnsupportedHeaderValue", + "message": "unsupported value in CBOR header" }, "note": "array headerValue 29" }, @@ -485,7 +538,8 @@ "hex": "9e", "expect": { "ok": false, - "code": "UnsupportedHeaderValue" + "code": "UnsupportedHeaderValue", + "message": "unsupported value in CBOR header" }, "note": "array headerValue 30" }, @@ -494,7 +548,8 @@ "hex": "bc", "expect": { "ok": false, - "code": "UnsupportedHeaderValue" + "code": "UnsupportedHeaderValue", + "message": "unsupported value in CBOR header" }, "note": "map headerValue 28" }, @@ -503,7 +558,8 @@ "hex": "bd", "expect": { "ok": false, - "code": "UnsupportedHeaderValue" + "code": "UnsupportedHeaderValue", + "message": "unsupported value in CBOR header" }, "note": "map headerValue 29" }, @@ -512,7 +568,8 @@ "hex": "be", "expect": { "ok": false, - "code": "UnsupportedHeaderValue" + "code": "UnsupportedHeaderValue", + "message": "unsupported value in CBOR header" }, "note": "map headerValue 30" }, @@ -521,7 +578,8 @@ "hex": "dc", "expect": { "ok": false, - "code": "UnsupportedHeaderValue" + "code": "UnsupportedHeaderValue", + "message": "unsupported value in CBOR header" }, "note": "tag headerValue 28" }, @@ -530,7 +588,8 @@ "hex": "dd", "expect": { "ok": false, - "code": "UnsupportedHeaderValue" + "code": "UnsupportedHeaderValue", + "message": "unsupported value in CBOR header" }, "note": "tag headerValue 29" }, @@ -539,7 +598,8 @@ "hex": "de", "expect": { "ok": false, - "code": "UnsupportedHeaderValue" + "code": "UnsupportedHeaderValue", + "message": "unsupported value in CBOR header" }, "note": "tag headerValue 30" }, @@ -548,7 +608,8 @@ "hex": "ff", "expect": { "ok": false, - "code": "UnsupportedHeaderValue" + "code": "UnsupportedHeaderValue", + "message": "unsupported value in CBOR header" }, "note": "standalone break = major 7 headerValue 31" }, @@ -557,16 +618,18 @@ "hex": "1800", "expect": { "ok": false, - "code": "NonCanonicalNumeric" + "code": "NonCanonicalNumeric", + "message": "a CBOR numeric value was encoded in non-canonical form" }, - "note": "uint 0 in 2-byte head (value<24, L107)" + "note": "uint 0 in 2-byte head (value<24)" }, { "name": "reject/NonCanonicalNumeric/1817", "hex": "1817", "expect": { "ok": false, - "code": "NonCanonicalNumeric" + "code": "NonCanonicalNumeric", + "message": "a CBOR numeric value was encoded in non-canonical form" }, "note": "uint 23 in 2-byte head - just below the 24 cutoff" }, @@ -575,16 +638,18 @@ "hex": "190017", "expect": { "ok": false, - "code": "NonCanonicalNumeric" + "code": "NonCanonicalNumeric", + "message": "a CBOR numeric value was encoded in non-canonical form" }, - "note": "uint 23 in 3-byte head (value<=0xff, L117)" + "note": "uint 23 in 3-byte head (value<=0xff)" }, { "name": "reject/NonCanonicalNumeric/1900ff", "hex": "1900ff", "expect": { "ok": false, - "code": "NonCanonicalNumeric" + "code": "NonCanonicalNumeric", + "message": "a CBOR numeric value was encoded in non-canonical form" }, "note": "uint 255 in 3-byte head - exactly at the <=0xff boundary" }, @@ -593,16 +658,18 @@ "hex": "1a00000017", "expect": { "ok": false, - "code": "NonCanonicalNumeric" + "code": "NonCanonicalNumeric", + "message": "a CBOR numeric value was encoded in non-canonical form" }, - "note": "uint 23 in 5-byte head (value<=0xffff, L129)" + "note": "uint 23 in 5-byte head (value<=0xffff)" }, { "name": "reject/NonCanonicalNumeric/1a0000ffff", "hex": "1a0000ffff", "expect": { "ok": false, - "code": "NonCanonicalNumeric" + "code": "NonCanonicalNumeric", + "message": "a CBOR numeric value was encoded in non-canonical form" }, "note": "uint 65535 in 5-byte head - exactly at the <=0xffff boundary" }, @@ -611,16 +678,18 @@ "hex": "1b0000000000000017", "expect": { "ok": false, - "code": "NonCanonicalNumeric" + "code": "NonCanonicalNumeric", + "message": "a CBOR numeric value was encoded in non-canonical form" }, - "note": "uint 23 in 9-byte head (value<=0xffffffff, L147)" + "note": "uint 23 in 9-byte head (value<=0xffffffff)" }, { "name": "reject/NonCanonicalNumeric/1b00000000ffffffff", "hex": "1b00000000ffffffff", "expect": { "ok": false, - "code": "NonCanonicalNumeric" + "code": "NonCanonicalNumeric", + "message": "a CBOR numeric value was encoded in non-canonical form" }, "note": "uint 2^32-1 in 9-byte head - exactly at the <=0xffffffff boundary" }, @@ -629,7 +698,8 @@ "hex": "3817", "expect": { "ok": false, - "code": "NonCanonicalNumeric" + "code": "NonCanonicalNumeric", + "message": "a CBOR numeric value was encoded in non-canonical form" }, "note": "negative -24 in 2-byte head (same minimality rule, major 1)" }, @@ -638,7 +708,8 @@ "hex": "390017", "expect": { "ok": false, - "code": "NonCanonicalNumeric" + "code": "NonCanonicalNumeric", + "message": "a CBOR numeric value was encoded in non-canonical form" }, "note": "negative 3-byte head non-minimal" }, @@ -647,7 +718,8 @@ "hex": "3900ff", "expect": { "ok": false, - "code": "NonCanonicalNumeric" + "code": "NonCanonicalNumeric", + "message": "a CBOR numeric value was encoded in non-canonical form" }, "note": "negative 3-byte head at <=0xff boundary" }, @@ -656,7 +728,8 @@ "hex": "3a00000017", "expect": { "ok": false, - "code": "NonCanonicalNumeric" + "code": "NonCanonicalNumeric", + "message": "a CBOR numeric value was encoded in non-canonical form" }, "note": "negative 5-byte head non-minimal" }, @@ -665,7 +738,8 @@ "hex": "3a0000ffff", "expect": { "ok": false, - "code": "NonCanonicalNumeric" + "code": "NonCanonicalNumeric", + "message": "a CBOR numeric value was encoded in non-canonical form" }, "note": "negative 5-byte head at <=0xffff boundary" }, @@ -674,7 +748,8 @@ "hex": "3b0000000000000017", "expect": { "ok": false, - "code": "NonCanonicalNumeric" + "code": "NonCanonicalNumeric", + "message": "a CBOR numeric value was encoded in non-canonical form" }, "note": "negative 9-byte head non-minimal" }, @@ -683,7 +758,8 @@ "hex": "3b00000000ffffffff", "expect": { "ok": false, - "code": "NonCanonicalNumeric" + "code": "NonCanonicalNumeric", + "message": "a CBOR numeric value was encoded in non-canonical form" }, "note": "negative 9-byte head at <=0xffffffff boundary" }, @@ -692,7 +768,8 @@ "hex": "5817", "expect": { "ok": false, - "code": "NonCanonicalNumeric" + "code": "NonCanonicalNumeric", + "message": "a CBOR numeric value was encoded in non-canonical form" }, "note": "bytestring LENGTH 23 in 2-byte head - same validation applies to length heads; fires before body underrun" }, @@ -701,7 +778,8 @@ "hex": "590001ff", "expect": { "ok": false, - "code": "NonCanonicalNumeric" + "code": "NonCanonicalNumeric", + "message": "a CBOR numeric value was encoded in non-canonical form" }, "note": "bytestring length 1 in 3-byte head (body byte present but head check fires first)" }, @@ -710,7 +788,8 @@ "hex": "7817", "expect": { "ok": false, - "code": "NonCanonicalNumeric" + "code": "NonCanonicalNumeric", + "message": "a CBOR numeric value was encoded in non-canonical form" }, "note": "text length 23 in 2-byte head" }, @@ -719,7 +798,8 @@ "hex": "9800", "expect": { "ok": false, - "code": "NonCanonicalNumeric" + "code": "NonCanonicalNumeric", + "message": "a CBOR numeric value was encoded in non-canonical form" }, "note": "array count 0 in 2-byte head" }, @@ -728,7 +808,8 @@ "hex": "b800", "expect": { "ok": false, - "code": "NonCanonicalNumeric" + "code": "NonCanonicalNumeric", + "message": "a CBOR numeric value was encoded in non-canonical form" }, "note": "map count 0 in 2-byte head" }, @@ -737,7 +818,8 @@ "hex": "d800", "expect": { "ok": false, - "code": "NonCanonicalNumeric" + "code": "NonCanonicalNumeric", + "message": "a CBOR numeric value was encoded in non-canonical form" }, "note": "tag 0 in 2-byte head" }, @@ -746,7 +828,8 @@ "hex": "d9001800", "expect": { "ok": false, - "code": "NonCanonicalNumeric" + "code": "NonCanonicalNumeric", + "message": "a CBOR numeric value was encoded in non-canonical form" }, "note": "tag 24 in 3-byte head (should be d818)" }, @@ -755,7 +838,8 @@ "hex": "f800", "expect": { "ok": false, - "code": "NonCanonicalNumeric" + "code": "NonCanonicalNumeric", + "message": "a CBOR numeric value was encoded in non-canonical form" }, "note": "simple 0 via 2-byte head: head-minimality check in readHeaderVarint precedes simple-value validation" }, @@ -764,7 +848,8 @@ "hex": "f814", "expect": { "ok": false, - "code": "NonCanonicalNumeric" + "code": "NonCanonicalNumeric", + "message": "a CBOR numeric value was encoded in non-canonical form" }, "note": "ORDER PROOF: 'false' (20) via 2-byte head is NonCanonicalNumeric, NOT InvalidSimpleValue" }, @@ -773,7 +858,8 @@ "hex": "f817", "expect": { "ok": false, - "code": "NonCanonicalNumeric" + "code": "NonCanonicalNumeric", + "message": "a CBOR numeric value was encoded in non-canonical form" }, "note": "simple 23 via 2-byte head - top of the f800-f817 NonCanonicalNumeric band" }, @@ -782,16 +868,18 @@ "hex": "fb4045000000000000", "expect": { "ok": false, - "code": "NonCanonicalNumeric" + "code": "NonCanonicalNumeric", + "message": "a CBOR numeric value was encoded in non-canonical form" }, - "note": "42.0 as f64: re-encode reduces to int 182a (checkCanonicalEncoding L306)" + "note": "42.0 as f64: re-encode reduces to int 182a (checkCanonicalEncoding)" }, { "name": "reject/NonCanonicalNumeric/fa42280000", "hex": "fa42280000", "expect": { "ok": false, - "code": "NonCanonicalNumeric" + "code": "NonCanonicalNumeric", + "message": "a CBOR numeric value was encoded in non-canonical form" }, "note": "42.0 as f32 reduces to int" }, @@ -800,7 +888,8 @@ "hex": "f95140", "expect": { "ok": false, - "code": "NonCanonicalNumeric" + "code": "NonCanonicalNumeric", + "message": "a CBOR numeric value was encoded in non-canonical form" }, "note": "42.0 as f16 reduces to int - integral floats reduce even at the smallest width" }, @@ -809,7 +898,8 @@ "hex": "f94200", "expect": { "ok": false, - "code": "NonCanonicalNumeric" + "code": "NonCanonicalNumeric", + "message": "a CBOR numeric value was encoded in non-canonical form" }, "note": "3.0 as f16 reduces to int 03" }, @@ -818,7 +908,8 @@ "hex": "f93c00", "expect": { "ok": false, - "code": "NonCanonicalNumeric" + "code": "NonCanonicalNumeric", + "message": "a CBOR numeric value was encoded in non-canonical form" }, "note": "1.0 as f16 reduces to int 01" }, @@ -827,7 +918,8 @@ "hex": "f9bc00", "expect": { "ok": false, - "code": "NonCanonicalNumeric" + "code": "NonCanonicalNumeric", + "message": "a CBOR numeric value was encoded in non-canonical form" }, "note": "-1.0 as f16 reduces to negative int 20" }, @@ -836,7 +928,8 @@ "hex": "fac0a00000", "expect": { "ok": false, - "code": "NonCanonicalNumeric" + "code": "NonCanonicalNumeric", + "message": "a CBOR numeric value was encoded in non-canonical form" }, "note": "-5.0 as f32 reduces to negative int 24" }, @@ -845,7 +938,8 @@ "hex": "fbc000000000000000", "expect": { "ok": false, - "code": "NonCanonicalNumeric" + "code": "NonCanonicalNumeric", + "message": "a CBOR numeric value was encoded in non-canonical form" }, "note": "-2.0 as f64 reduces to negative int 21" }, @@ -854,7 +948,8 @@ "hex": "fa47c35000", "expect": { "ok": false, - "code": "NonCanonicalNumeric" + "code": "NonCanonicalNumeric", + "message": "a CBOR numeric value was encoded in non-canonical form" }, "note": "100000.0 as f32 reduces to int 1a000186a0" }, @@ -863,16 +958,48 @@ "hex": "fb43e0000000000000", "expect": { "ok": false, - "code": "NonCanonicalNumeric" + "code": "NonCanonicalNumeric", + "message": "a CBOR numeric value was encoded in non-canonical form" }, "note": "2^63 as f64 reduces to uint 1b8000000000000000 (bigint reduction path)" }, + { + "name": "reject/NonCanonicalNumeric/fa4f000000", + "hex": "fa4f000000", + "expect": { + "ok": false, + "code": "NonCanonicalNumeric", + "message": "a CBOR numeric value was encoded in non-canonical form" + }, + "note": "2^31 as f32: `as i32` saturates to i32::MAX, which rounds back to exactly 2^31 -> whole -> rejected" + }, + { + "name": "reject/NonCanonicalNumeric/facf000000", + "hex": "facf000000", + "expect": { + "ok": false, + "code": "NonCanonicalNumeric", + "message": "a CBOR numeric value was encoded in non-canonical form" + }, + "note": "-2^31 as f32: i32::MIN round-trips exactly -> rejected" + }, + { + "name": "reject/NonCanonicalNumeric/fbc3e0000000000000", + "hex": "fbc3e0000000000000", + "expect": { + "ok": false, + "code": "NonCanonicalNumeric", + "message": "a CBOR numeric value was encoded in non-canonical form" + }, + "note": "-2^63 as f64: i64::MIN round-trips exactly -> rejected" + }, { "name": "reject/NonCanonicalNumeric/fa3fc00000", "hex": "fa3fc00000", "expect": { "ok": false, - "code": "NonCanonicalNumeric" + "code": "NonCanonicalNumeric", + "message": "a CBOR numeric value was encoded in non-canonical form" }, "note": "1.5 as f32 when f16 f93e00 suffices" }, @@ -881,7 +1008,8 @@ "hex": "fb3ff8000000000000", "expect": { "ok": false, - "code": "NonCanonicalNumeric" + "code": "NonCanonicalNumeric", + "message": "a CBOR numeric value was encoded in non-canonical form" }, "note": "1.5 as f64 when f16 suffices" }, @@ -890,7 +1018,8 @@ "hex": "fa3f000000", "expect": { "ok": false, - "code": "NonCanonicalNumeric" + "code": "NonCanonicalNumeric", + "message": "a CBOR numeric value was encoded in non-canonical form" }, "note": "0.5 as f32 when f16 f93800 suffices" }, @@ -899,7 +1028,8 @@ "hex": "fb3fe0000000000000", "expect": { "ok": false, - "code": "NonCanonicalNumeric" + "code": "NonCanonicalNumeric", + "message": "a CBOR numeric value was encoded in non-canonical form" }, "note": "0.5 as f64 when f16 suffices" }, @@ -908,7 +1038,8 @@ "hex": "fb3fb99999a0000000", "expect": { "ok": false, - "code": "NonCanonicalNumeric" + "code": "NonCanonicalNumeric", + "message": "a CBOR numeric value was encoded in non-canonical form" }, "note": "0.10000000149011612 (exactly f32 0x3dcccccd) widened to f64 - must be fa3dcccccd" }, @@ -917,7 +1048,8 @@ "hex": "fb7ff8000000000000", "expect": { "ok": false, - "code": "NonCanonicalNumeric" + "code": "NonCanonicalNumeric", + "message": "a CBOR numeric value was encoded in non-canonical form" }, "note": "f64 quiet NaN: every NaN must be canonical f97e00" }, @@ -926,7 +1058,8 @@ "hex": "fa7fc00000", "expect": { "ok": false, - "code": "NonCanonicalNumeric" + "code": "NonCanonicalNumeric", + "message": "a CBOR numeric value was encoded in non-canonical form" }, "note": "f32 quiet NaN -> f97e00" }, @@ -935,7 +1068,8 @@ "hex": "f97e01", "expect": { "ok": false, - "code": "NonCanonicalNumeric" + "code": "NonCanonicalNumeric", + "message": "a CBOR numeric value was encoded in non-canonical form" }, "note": "f16 NaN with nonzero payload" }, @@ -944,7 +1078,8 @@ "hex": "f97c01", "expect": { "ok": false, - "code": "NonCanonicalNumeric" + "code": "NonCanonicalNumeric", + "message": "a CBOR numeric value was encoded in non-canonical form" }, "note": "f16 signalling NaN" }, @@ -953,7 +1088,8 @@ "hex": "f9fc01", "expect": { "ok": false, - "code": "NonCanonicalNumeric" + "code": "NonCanonicalNumeric", + "message": "a CBOR numeric value was encoded in non-canonical form" }, "note": "f16 negative NaN" }, @@ -962,7 +1098,8 @@ "hex": "fb7ff0000000000001", "expect": { "ok": false, - "code": "NonCanonicalNumeric" + "code": "NonCanonicalNumeric", + "message": "a CBOR numeric value was encoded in non-canonical form" }, "note": "f64 signalling NaN payload" }, @@ -971,7 +1108,8 @@ "hex": "fa7f800000", "expect": { "ok": false, - "code": "NonCanonicalNumeric" + "code": "NonCanonicalNumeric", + "message": "a CBOR numeric value was encoded in non-canonical form" }, "note": "+Infinity as f32: canonical form is f97c00" }, @@ -980,7 +1118,8 @@ "hex": "faff800000", "expect": { "ok": false, - "code": "NonCanonicalNumeric" + "code": "NonCanonicalNumeric", + "message": "a CBOR numeric value was encoded in non-canonical form" }, "note": "-Infinity as f32: canonical form is f9fc00" }, @@ -989,7 +1128,8 @@ "hex": "fb7ff0000000000000", "expect": { "ok": false, - "code": "NonCanonicalNumeric" + "code": "NonCanonicalNumeric", + "message": "a CBOR numeric value was encoded in non-canonical form" }, "note": "+Infinity as f64" }, @@ -998,7 +1138,8 @@ "hex": "fbfff0000000000000", "expect": { "ok": false, - "code": "NonCanonicalNumeric" + "code": "NonCanonicalNumeric", + "message": "a CBOR numeric value was encoded in non-canonical form" }, "note": "-Infinity as f64" }, @@ -1007,7 +1148,8 @@ "hex": "f98000", "expect": { "ok": false, - "code": "NonCanonicalNumeric" + "code": "NonCanonicalNumeric", + "message": "a CBOR numeric value was encoded in non-canonical form" }, "note": "-0.0 as f16: dCBOR reduces -0.0 to integer 00" }, @@ -1016,7 +1158,8 @@ "hex": "fa80000000", "expect": { "ok": false, - "code": "NonCanonicalNumeric" + "code": "NonCanonicalNumeric", + "message": "a CBOR numeric value was encoded in non-canonical form" }, "note": "-0.0 as f32 -> integer 00" }, @@ -1025,7 +1168,8 @@ "hex": "fb8000000000000000", "expect": { "ok": false, - "code": "NonCanonicalNumeric" + "code": "NonCanonicalNumeric", + "message": "a CBOR numeric value was encoded in non-canonical form" }, "note": "-0.0 as f64 -> integer 00" }, @@ -1034,7 +1178,8 @@ "hex": "f90000", "expect": { "ok": false, - "code": "NonCanonicalNumeric" + "code": "NonCanonicalNumeric", + "message": "a CBOR numeric value was encoded in non-canonical form" }, "note": "+0.0 as f16 -> integer 00 (all zero floats reduce)" }, @@ -1043,7 +1188,8 @@ "hex": "f97bff", "expect": { "ok": false, - "code": "NonCanonicalNumeric" + "code": "NonCanonicalNumeric", + "message": "a CBOR numeric value was encoded in non-canonical form" }, "note": "TRAP: f16 max 65504 is integral, reduces to 19ffe0 - rejected, not a valid f16" }, @@ -1052,7 +1198,8 @@ "hex": "fa33800000", "expect": { "ok": false, - "code": "NonCanonicalNumeric" + "code": "NonCanonicalNumeric", + "message": "a CBOR numeric value was encoded in non-canonical form" }, "note": "5.9604644775390625e-8 as f32 when the f16 subnormal f90001 suffices" }, @@ -1061,7 +1208,8 @@ "hex": "e0", "expect": { "ok": false, - "code": "InvalidSimpleValue" + "code": "InvalidSimpleValue", + "message": "an invalid CBOR simple value was encountered" }, "note": "simple 0 (major 7 immediate, not false/true/null/float)" }, @@ -1070,7 +1218,8 @@ "hex": "ef", "expect": { "ok": false, - "code": "InvalidSimpleValue" + "code": "InvalidSimpleValue", + "message": "an invalid CBOR simple value was encountered" }, "note": "simple 15" }, @@ -1079,7 +1228,8 @@ "hex": "f0", "expect": { "ok": false, - "code": "InvalidSimpleValue" + "code": "InvalidSimpleValue", + "message": "an invalid CBOR simple value was encountered" }, "note": "simple 16" }, @@ -1088,7 +1238,8 @@ "hex": "f3", "expect": { "ok": false, - "code": "InvalidSimpleValue" + "code": "InvalidSimpleValue", + "message": "an invalid CBOR simple value was encountered" }, "note": "simple 19 (last before false)" }, @@ -1097,7 +1248,8 @@ "hex": "f7", "expect": { "ok": false, - "code": "InvalidSimpleValue" + "code": "InvalidSimpleValue", + "message": "an invalid CBOR simple value was encountered" }, "note": "undefined (simple 23) - dCBOR forbids it" }, @@ -1106,7 +1258,8 @@ "hex": "f818", "expect": { "ok": false, - "code": "InvalidSimpleValue" + "code": "InvalidSimpleValue", + "message": "an invalid CBOR simple value was encountered" }, "note": "simple 24: passes head-minimality (24>=24), then fails simple dispatch - first 2-byte InvalidSimpleValue" }, @@ -1115,7 +1268,8 @@ "hex": "f820", "expect": { "ok": false, - "code": "InvalidSimpleValue" + "code": "InvalidSimpleValue", + "message": "an invalid CBOR simple value was encountered" }, "note": "simple 32" }, @@ -1124,7 +1278,8 @@ "hex": "f8ff", "expect": { "ok": false, - "code": "InvalidSimpleValue" + "code": "InvalidSimpleValue", + "message": "an invalid CBOR simple value was encountered" }, "note": "simple 255" }, @@ -1133,7 +1288,8 @@ "hex": "62c328", "expect": { "ok": false, - "code": "InvalidUtf8" + "code": "InvalidUtf8", + "message": "invalid UTF‑8 string: invalid utf-8 sequence of 1 bytes from index 0" }, "note": "0xc3 lead byte followed by invalid continuation 0x28" }, @@ -1142,7 +1298,8 @@ "hex": "62c080", "expect": { "ok": false, - "code": "InvalidUtf8" + "code": "InvalidUtf8", + "message": "invalid UTF‑8 string: invalid utf-8 sequence of 1 bytes from index 0" }, "note": "overlong 2-byte encoding of NUL" }, @@ -1151,7 +1308,8 @@ "hex": "63eda080", "expect": { "ok": false, - "code": "InvalidUtf8" + "code": "InvalidUtf8", + "message": "invalid UTF‑8 string: invalid utf-8 sequence of 1 bytes from index 0" }, "note": "CESU-8 lone surrogate U+D800 (ed a0 80)" }, @@ -1160,7 +1318,8 @@ "hex": "61c3", "expect": { "ok": false, - "code": "InvalidUtf8" + "code": "InvalidUtf8", + "message": "invalid UTF‑8 string: incomplete utf-8 byte sequence from index 0" }, "note": "declared length covers only the lead byte of a 2-byte sequence" }, @@ -1169,7 +1328,8 @@ "hex": "64f4908080", "expect": { "ok": false, - "code": "InvalidUtf8" + "code": "InvalidUtf8", + "message": "invalid UTF‑8 string: invalid utf-8 sequence of 1 bytes from index 0" }, "note": "codepoint U+110000 above Unicode max" }, @@ -1178,16 +1338,98 @@ "hex": "61ff", "expect": { "ok": false, - "code": "InvalidUtf8" + "code": "InvalidUtf8", + "message": "invalid UTF‑8 string: invalid utf-8 sequence of 1 bytes from index 0" }, "note": "0xff is never valid in UTF-8" }, + { + "name": "reject/InvalidUtf8/63e28228", + "hex": "63e28228", + "expect": { + "ok": false, + "code": "InvalidUtf8", + "message": "invalid UTF‑8 string: invalid utf-8 sequence of 2 bytes from index 0" + }, + "note": "3-byte lead, valid second byte, invalid third: error_len 2" + }, + { + "name": "reject/InvalidUtf8/64f09f9841", + "hex": "64f09f9841", + "expect": { + "ok": false, + "code": "InvalidUtf8", + "message": "invalid UTF‑8 string: invalid utf-8 sequence of 3 bytes from index 0" + }, + "note": "4-byte lead, two valid continuations, invalid fourth: error_len 3" + }, + { + "name": "reject/InvalidUtf8/63e0a041", + "hex": "63e0a041", + "expect": { + "ok": false, + "code": "InvalidUtf8", + "message": "invalid UTF‑8 string: invalid utf-8 sequence of 2 bytes from index 0" + }, + "note": "e0 with valid second byte a0, invalid third: error_len 2" + }, + { + "name": "reject/InvalidUtf8/62f0a0", + "hex": "62f0a0", + "expect": { + "ok": false, + "code": "InvalidUtf8", + "message": "invalid UTF‑8 string: incomplete utf-8 byte sequence from index 0" + }, + "note": "4-byte sequence cut after a valid second byte: incomplete" + }, + { + "name": "reject/InvalidUtf8/6441c3a9ff", + "hex": "6441c3a9ff", + "expect": { + "ok": false, + "code": "InvalidUtf8", + "message": "invalid UTF‑8 string: invalid utf-8 sequence of 1 bytes from index 3" + }, + "note": "valid 'Aé' then ff: valid_up_to 3" + }, + { + "name": "reject/InvalidUtf8/64616263ff", + "hex": "64616263ff", + "expect": { + "ok": false, + "code": "InvalidUtf8", + "message": "invalid UTF‑8 string: invalid utf-8 sequence of 1 bytes from index 3" + }, + "note": "valid 'abc' then ff: valid_up_to 3" + }, + { + "name": "reject/InvalidUtf8/62eda0", + "hex": "62eda0", + "expect": { + "ok": false, + "code": "InvalidUtf8", + "message": "invalid UTF‑8 string: invalid utf-8 sequence of 1 bytes from index 0" + }, + "note": "ed with second byte a0 (surrogate range) and no third: the present bad byte beats incomplete" + }, + { + "name": "reject/InvalidUtf8/62c0af", + "hex": "62c0af", + "expect": { + "ok": false, + "code": "InvalidUtf8", + "message": "invalid UTF‑8 string: invalid utf-8 sequence of 1 bytes from index 0" + }, + "note": "overlong lead c0: error_len 1 whatever follows" + }, { "name": "reject/NonCanonicalString/6365cc81", "hex": "6365cc81", "expect": { "ok": false, - "code": "NonCanonicalString" + "code": "NonCanonicalString", + "message": "a CBOR string was not encoded in Unicode Canonical Normalization Form C" }, "note": "'e' + combining acute U+0301 (NFD) - NFC composes to U+00E9" }, @@ -1196,7 +1438,8 @@ "hex": "66e18480e185a1", "expect": { "ok": false, - "code": "NonCanonicalString" + "code": "NonCanonicalString", + "message": "a CBOR string was not encoded in Unicode Canonical Normalization Form C" }, "note": "decomposed Hangul jamo U+1100 U+1161 - NFC composes to U+AC00" }, @@ -1205,7 +1448,8 @@ "hex": "63e284ab", "expect": { "ok": false, - "code": "NonCanonicalString" + "code": "NonCanonicalString", + "message": "a CBOR string was not encoded in Unicode Canonical Normalization Form C" }, "note": "U+212B ANGSTROM SIGN - NFC singleton maps to U+00C5" }, @@ -1214,7 +1458,8 @@ "hex": "0000", "expect": { "ok": false, - "code": "UnusedData" + "code": "UnusedData", + "message": "the decoded CBOR had 1 extra bytes at the end" }, "note": "uint 0 decoded, 1 trailing byte (details.count=1)" }, @@ -1223,7 +1468,8 @@ "hex": "f6f6", "expect": { "ok": false, - "code": "UnusedData" + "code": "UnusedData", + "message": "the decoded CBOR had 1 extra bytes at the end" }, "note": "null decoded, trailing null byte" }, @@ -1232,7 +1478,8 @@ "hex": "4100ff", "expect": { "ok": false, - "code": "UnusedData" + "code": "UnusedData", + "message": "the decoded CBOR had 1 extra bytes at the end" }, "note": "complete 1-byte bytestring + trailing 0xff" }, @@ -1241,7 +1488,8 @@ "hex": "a201010102", "expect": { "ok": false, - "code": "DuplicateMapKey" + "code": "DuplicateMapKey", + "message": "the decoded CBOR map has a duplicate key" }, "note": "map {1:1, 1:2}: second key byte-equal to max - has() check fires BEFORE order check" }, @@ -1250,7 +1498,8 @@ "hex": "a202000100", "expect": { "ok": false, - "code": "MisorderedMapKey" + "code": "MisorderedMapKey", + "message": "the decoded CBOR map has keys that are not in canonical order" }, "note": "same-length int keys descending (2 then 1)" }, @@ -1259,7 +1508,8 @@ "hex": "a21818001700", "expect": { "ok": false, - "code": "MisorderedMapKey" + "code": "MisorderedMapKey", + "message": "the decoded CBOR map has keys that are not in canonical order" }, "note": "key 24 (2-byte 1818) then key 23 (1-byte 17): shorter/smaller encoding must sort first (0x17<0x18)" }, @@ -1268,7 +1518,8 @@ "hex": "a21818000100", "expect": { "ok": false, - "code": "MisorderedMapKey" + "code": "MisorderedMapKey", + "message": "the decoded CBOR map has keys that are not in canonical order" }, "note": "2-byte key head 1818 then 1-byte key 01" }, @@ -1277,7 +1528,8 @@ "hex": "a26161000100", "expect": { "ok": false, - "code": "MisorderedMapKey" + "code": "MisorderedMapKey", + "message": "the decoded CBOR map has keys that are not in canonical order" }, "note": "text key \"a\" (6161) then int key 1 (01): int head byte sorts before text head" }, @@ -1286,7 +1538,8 @@ "hex": "a3010002000100", "expect": { "ok": false, - "code": "DuplicateMapKey" + "code": "DuplicateMapKey", + "message": "the decoded CBOR map has a duplicate key" }, "note": "3-entry map: third key duplicates a NON-max earlier key (1) - still DuplicateMapKey, not Misordered" }, @@ -1295,7 +1548,8 @@ "hex": "5b0020000000000000", "expect": { "ok": false, - "code": "Underrun" + "code": "Underrun", + "message": "early end of CBOR data" }, "note": "bytestring length 2^53: narrowInteger keeps it bigint (> MAX_SAFE_INTEGER); unsatisfiable body -> Underrun, matching Rust's usize bounds check" }, @@ -1304,7 +1558,8 @@ "hex": "5bffffffffffffffff", "expect": { "ok": false, - "code": "Underrun" + "code": "Underrun", + "message": "early end of CBOR data" }, "note": "bytestring length 2^64-1 (bigint)" }, @@ -1313,7 +1568,8 @@ "hex": "7b0020000000000000", "expect": { "ok": false, - "code": "Underrun" + "code": "Underrun", + "message": "early end of CBOR data" }, "note": "text length 2^53 (bigint)" }, @@ -1322,7 +1578,8 @@ "hex": "7bffffffffffffffff", "expect": { "ok": false, - "code": "Underrun" + "code": "Underrun", + "message": "early end of CBOR data" }, "note": "text length 2^64-1 (bigint)" }, @@ -1331,7 +1588,8 @@ "hex": "811817", "expect": { "ok": false, - "code": "NonCanonicalNumeric" + "code": "NonCanonicalNumeric", + "message": "a CBOR numeric value was encoded in non-canonical form" }, "note": "nested: non-minimal int head inside array" }, @@ -1340,7 +1598,8 @@ "hex": "81f7", "expect": { "ok": false, - "code": "InvalidSimpleValue" + "code": "InvalidSimpleValue", + "message": "an invalid CBOR simple value was encountered" }, "note": "nested: undefined inside array" }, @@ -1349,7 +1608,8 @@ "hex": "811c", "expect": { "ok": false, - "code": "UnsupportedHeaderValue" + "code": "UnsupportedHeaderValue", + "message": "unsupported value in CBOR header" }, "note": "nested: headerValue 28 inside array" }, @@ -1358,7 +1618,8 @@ "hex": "8161", "expect": { "ok": false, - "code": "Underrun" + "code": "Underrun", + "message": "early end of CBOR data" }, "note": "nested: truncated text body inside array" }, @@ -1367,7 +1628,8 @@ "hex": "8162c328", "expect": { "ok": false, - "code": "InvalidUtf8" + "code": "InvalidUtf8", + "message": "invalid UTF‑8 string: invalid utf-8 sequence of 1 bytes from index 0" }, "note": "nested: invalid UTF-8 inside array" }, @@ -1376,7 +1638,8 @@ "hex": "81a201010102", "expect": { "ok": false, - "code": "DuplicateMapKey" + "code": "DuplicateMapKey", + "message": "the decoded CBOR map has a duplicate key" }, "note": "nested: duplicate-key map inside array" }, @@ -1385,7 +1648,8 @@ "hex": "81a202000100", "expect": { "ok": false, - "code": "MisorderedMapKey" + "code": "MisorderedMapKey", + "message": "the decoded CBOR map has keys that are not in canonical order" }, "note": "nested: misordered map inside array" }, @@ -1394,7 +1658,8 @@ "hex": "81fb4045000000000000", "expect": { "ok": false, - "code": "NonCanonicalNumeric" + "code": "NonCanonicalNumeric", + "message": "a CBOR numeric value was encoded in non-canonical form" }, "note": "nested: integer-reducible f64 inside array" }, @@ -1403,7 +1668,8 @@ "hex": "82005b0020000000000000", "expect": { "ok": false, - "code": "Underrun" + "code": "Underrun", + "message": "early end of CBOR data" }, "note": "nested: bigint bytestring length as 2nd array element" }, @@ -1412,7 +1678,8 @@ "hex": "a1001817", "expect": { "ok": false, - "code": "NonCanonicalNumeric" + "code": "NonCanonicalNumeric", + "message": "a CBOR numeric value was encoded in non-canonical form" }, "note": "nested: non-minimal int as map VALUE" }, @@ -1421,7 +1688,8 @@ "hex": "a100f7", "expect": { "ok": false, - "code": "InvalidSimpleValue" + "code": "InvalidSimpleValue", + "message": "an invalid CBOR simple value was encountered" }, "note": "nested: undefined as map value" }, @@ -1430,7 +1698,8 @@ "hex": "a1f700", "expect": { "ok": false, - "code": "InvalidSimpleValue" + "code": "InvalidSimpleValue", + "message": "an invalid CBOR simple value was encountered" }, "note": "nested: undefined as map KEY (key decode fails before setNext)" }, @@ -1439,7 +1708,8 @@ "hex": "a1181700", "expect": { "ok": false, - "code": "NonCanonicalNumeric" + "code": "NonCanonicalNumeric", + "message": "a CBOR numeric value was encoded in non-canonical form" }, "note": "nested: non-minimal int as map key" }, @@ -1448,7 +1718,8 @@ "hex": "a1005f", "expect": { "ok": false, - "code": "UnsupportedHeaderValue" + "code": "UnsupportedHeaderValue", + "message": "unsupported value in CBOR header" }, "note": "nested: indefinite bytestring as map value" }, @@ -1457,7 +1728,8 @@ "hex": "c11817", "expect": { "ok": false, - "code": "NonCanonicalNumeric" + "code": "NonCanonicalNumeric", + "message": "a CBOR numeric value was encoded in non-canonical form" }, "note": "nested: non-minimal int as tag content" }, @@ -1466,7 +1738,8 @@ "hex": "c1f7", "expect": { "ok": false, - "code": "InvalidSimpleValue" + "code": "InvalidSimpleValue", + "message": "an invalid CBOR simple value was encountered" }, "note": "nested: undefined as tag content" }, @@ -1475,7 +1748,8 @@ "hex": "c15f", "expect": { "ok": false, - "code": "UnsupportedHeaderValue" + "code": "UnsupportedHeaderValue", + "message": "unsupported value in CBOR header" }, "note": "nested: indefinite bytestring as tag content" }, @@ -1484,7 +1758,8 @@ "hex": "c162c328", "expect": { "ok": false, - "code": "InvalidUtf8" + "code": "InvalidUtf8", + "message": "invalid UTF‑8 string: invalid utf-8 sequence of 1 bytes from index 0" }, "note": "nested: invalid UTF-8 as tag content" }, @@ -1493,7 +1768,8 @@ "hex": "c1a202000100", "expect": { "ok": false, - "code": "MisorderedMapKey" + "code": "MisorderedMapKey", + "message": "the decoded CBOR map has keys that are not in canonical order" }, "note": "nested: misordered map inside tag" }, @@ -1502,7 +1778,8 @@ "hex": "c1f97e01", "expect": { "ok": false, - "code": "NonCanonicalNumeric" + "code": "NonCanonicalNumeric", + "message": "a CBOR numeric value was encoded in non-canonical form" }, "note": "nested: non-canonical NaN inside tag" }, @@ -1511,7 +1788,8 @@ "hex": "da000001", "expect": { "ok": false, - "code": "Underrun" + "code": "Underrun", + "message": "early end of CBOR data" }, "note": "tag 5-byte head, 3 of 4 arg bytes" }, @@ -1520,7 +1798,8 @@ "hex": "db00000000000001", "expect": { "ok": false, - "code": "Underrun" + "code": "Underrun", + "message": "early end of CBOR data" }, "note": "tag 9-byte head, 7 of 8 arg bytes" }, @@ -1529,7 +1808,8 @@ "hex": "da0000001800", "expect": { "ok": false, - "code": "NonCanonicalNumeric" + "code": "NonCanonicalNumeric", + "message": "a CBOR numeric value was encoded in non-canonical form" }, "note": "tag 24 in 5-byte head (should be d818)" }, @@ -1538,7 +1818,8 @@ "hex": "db000000000000001800", "expect": { "ok": false, - "code": "NonCanonicalNumeric" + "code": "NonCanonicalNumeric", + "message": "a CBOR numeric value was encoded in non-canonical form" }, "note": "tag 24 in 9-byte head" }, @@ -1547,7 +1828,8 @@ "hex": "da0000ffff00", "expect": { "ok": false, - "code": "NonCanonicalNumeric" + "code": "NonCanonicalNumeric", + "message": "a CBOR numeric value was encoded in non-canonical form" }, "note": "tag 65535 in 5-byte head at the <=0xffff boundary" }, @@ -1556,7 +1838,8 @@ "hex": "db00000000ffffffff00", "expect": { "ok": false, - "code": "NonCanonicalNumeric" + "code": "NonCanonicalNumeric", + "message": "a CBOR numeric value was encoded in non-canonical form" }, "note": "tag 2^32-1 in 9-byte head at the <=0xffffffff boundary" }, @@ -1565,7 +1848,8 @@ "hex": "7818", "expect": { "ok": false, - "code": "Underrun" + "code": "Underrun", + "message": "early end of CBOR data" }, "note": "text 2-byte length head declares 24 body bytes, has 0" }, @@ -1574,7 +1858,8 @@ "hex": "5901f4", "expect": { "ok": false, - "code": "Underrun" + "code": "Underrun", + "message": "early end of CBOR data" }, "note": "bytestring 3-byte length head truncated (1 of 2 length bytes)" }, @@ -1583,7 +1868,8 @@ "hex": "82f7f6", "expect": { "ok": false, - "code": "InvalidSimpleValue" + "code": "InvalidSimpleValue", + "message": "an invalid CBOR simple value was encountered" }, "note": "undefined as first array element, decode stops there" }, @@ -1592,7 +1878,8 @@ "hex": "a20102616162", "expect": { "ok": false, - "code": "Underrun" + "code": "Underrun", + "message": "early end of CBOR data" }, "note": "2-entry map second value is a text head with missing body" }, @@ -1828,6 +2115,109 @@ }, "note": "2^64 exactly as f32 - whole-valued but exceeds u64 by 1, so no integer reduction" }, + { + "name": "accept/fa4f000001-as-1a80000100", + "hex": "fa4f000001", + "expect": { + "ok": true, + "hex": "1a80000100" + }, + "note": "2^31+256: `as i32` saturates to 2^31 != value, so accepted; From -> Unsigned(2147483904)" + }, + { + "name": "accept/fa4f7fffff-as-1affffff00", + "hex": "fa4f7fffff", + "expect": { + "ok": true, + "hex": "1affffff00" + }, + "note": "2^32-256 (largest f32 below 2^32) -> Unsigned(4294967040)" + }, + { + "name": "accept/facf000001-as-3a80000100", + "hex": "facf000001", + "expect": { + "ok": true, + "hex": "3a80000100" + }, + "note": "-(2^31+256): -1f32 - n rounds to 2^31+256 -> Negative(2147483904) = -2147483905" + }, + { + "name": "accept/fadf000000-as-3b8000000000000000", + "hex": "fadf000000", + "expect": { + "ok": true, + "hex": "3b8000000000000000" + }, + "note": "-2^63 as f32: -1f32 - n rounds to 2^63 -> Negative(2^63) = -9223372036854775809 (65-bit)" + }, + { + "name": "accept/fadb000000-as-3b0080000000000000", + "hex": "fadb000000", + "expect": { + "ok": true, + "hex": "3b0080000000000000" + }, + "note": "-2^55 as f32: -1f32 - n rounds to 2^55 -> Negative(2^55) = -36028797018963969" + }, + { + "name": "accept/fb43e0000000000001-as-1b8000000000000800", + "hex": "fb43e0000000000001", + "expect": { + "ok": true, + "hex": "1b8000000000000800" + }, + "note": "2^63+2048: `as i64` saturates to i64::MAX -> 2^63 != value, accepted; From -> Unsigned" + }, + { + "name": "accept/fbc3e0000000000001-as-3b80000000000007ff", + "hex": "fbc3e0000000000001", + "expect": { + "ok": true, + "hex": "3b80000000000007ff" + }, + "note": "-(2^63+2048): accepted via i64 saturation; From -> Negative(2^63+2047)" + }, + { + "name": "accept/fa4f800000", + "hex": "fa4f800000", + "expect": { + "ok": true + }, + "note": "2^32 as f32: accepted (saturating i32 image differs); no u32 fits, so the node stays a float" + }, + { + "name": "accept/fa4fc00000", + "hex": "fa4fc00000", + "expect": { + "ok": true + }, + "note": "1.5*2^32 as f32: whole, accepted, stays a float" + }, + { + "name": "accept/fa5a000000", + "hex": "fa5a000000", + "expect": { + "ok": true + }, + "note": "2^53 as f32: whole, accepted, stays a float" + }, + { + "name": "accept/fa5f000000", + "hex": "fa5f000000", + "expect": { + "ok": true + }, + "note": "2^63 as f32: whole, accepted, stays a float" + }, + { + "name": "accept/fadf800000", + "hex": "fadf800000", + "expect": { + "ok": true + }, + "note": "-2^64 as f32: -1f32 - n rounds to 2^64, which no u64 holds, so the node stays a float" + }, { "name": "accept/fb3ff199999999999a", "hex": "fb3ff199999999999a", @@ -1948,6 +2338,30 @@ }, "note": "NFC-composed é (U+00E9) - passes both UTF-8 and NFC checks" }, + { + "name": "accept/64efbbbf61", + "hex": "64efbbbf61", + "expect": { + "ok": true + }, + "note": "text \"a\": a leading BOM is a character, kept on decode (Rust String::from_utf8 parity)" + }, + { + "name": "accept/63efbbbf", + "hex": "63efbbbf", + "expect": { + "ok": true + }, + "note": "text consisting of only U+FEFF - re-encodes to the same 3 bytes, not to the empty string" + }, + { + "name": "accept/8263efbbbf63efbbbf", + "hex": "8263efbbbf63efbbbf", + "expect": { + "ok": true + }, + "note": "two BOM-only strings in an array - each element keeps its BOM" + }, { "name": "accept/8181818100", "hex": "8181818100", diff --git a/tests/vectors/encode-corpus.ts b/tests/vectors/encode-corpus.ts index 4e3d0bf..0a4fa53 100644 --- a/tests/vectors/encode-corpus.ts +++ b/tests/vectors/encode-corpus.ts @@ -1,34 +1,29 @@ /** - * Curated golden ENCODE corpus (API_REDESIGN_PLAN P1.1a). + * Curated golden ENCODE corpus. * * Every entry is a named, build-agnostic construction recipe. The generator - * (`scripts/generate-vectors.mjs`) encodes each with the working tree and + * (`scripts/generate-vectors.ts`) encodes each with the working tree and * commits the expected outcome (hex, or CborError code for inputs that throw) * to `tests/vectors/encode-vectors.json`; `tests/golden-vectors.test.ts` * verifies the working tree against that fixture on every run. * - * Entries marked `tombstone` are the two input shapes the redesign will make - * throw (P3.5 `{tag,value}` sniffing, P3.7 `taggedCbor`-without-`toCbor` - * auto-wrap). They MUST keep their baseline bytes until the breaking wave - * lands; the golden test asserts them via the same fixture, and flipping them - * to expected-throw is a deliberate, reviewed fixture regeneration. + * Entries marked `tombstone` are the two input shapes that throw a directive + * `Custom` error: a plain `{tag, value}` object literal and an object with + * `taggedCbor()` but no `toCbor()`. The Rust harness skips them. * - * Sources for the boundary/quirk values: live-verified recon of - * src/float.ts + src/varint.ts + src/cbor.ts dispatch (see P1.1 notes in - * API_REDESIGN_PLAN.md §6). The three frozen encoder quirks deliberately - * covered: + * The encoder quirks deliberately covered: * Q1 f32 negative reduction uses Math.fround(-1-n) (Rust parity), which - * COLLIDES byte-wise for f32-exact negatives beyond 2^24 + * collides byte-wise for f32-exact negatives beyond 2^24 * (e.g. -16777218.0 encodes as semantic -16777217); - * Q2 f32-exact whole values >= 2^32 do NOT integer-reduce (stay 0xfa); + * Q2 f32-exact whole values >= 2^32 do not integer-reduce (stay 0xfa); * Q3 -0.0 encodes as integer 0x00 (sign lost). * Q1/Q2 live in the float encoder's own reduction ladder, which plain whole - * numbers NEVER reach (cbor() dispatch integer-reduces them exactly first) - + * numbers never reach (cbor() dispatch integer-reduces them exactly first) - * they are pinned via the bare-Float-node vectors in section 2b. Q3 is * visible through both routes. */ -import type { Recipe } from "./recipes"; +import type { Recipe, RemovedInputShape } from "./recipes"; // Terse constructors - keep the table readable. const n = (v: number | string): Recipe => ({ k: "n", v: String(v) }); @@ -67,11 +62,11 @@ const taggedproto = (tag: string | number, inner: Recipe): Recipe => ({ export interface EncodeCorpusEntry { name: string; recipe: Recipe; - /** Set on the two redesign tombstone shapes (P3.5 / P3.7). */ - tombstone?: "P3.5" | "P3.7"; + /** Set on the removed input shapes, which throw a directive error. */ + tombstone?: RemovedInputShape; } -const e = (name: string, recipe: Recipe, tombstone?: "P3.5" | "P3.7"): EncodeCorpusEntry => +const e = (name: string, recipe: Recipe, tombstone?: RemovedInputShape): EncodeCorpusEntry => tombstone === undefined ? { name, recipe } : { name, recipe, tombstone }; // --------------------------------------------------------------------------- @@ -143,7 +138,7 @@ const integers: EncodeCorpusEntry[] = [ ]; // --------------------------------------------------------------------------- -// 2. Floats - every f16/f32/f64 canonical edge and the three frozen quirks. +// 2. Floats - every f16/f32/f64 canonical edge and the three quirks. // (The full 77-value adversarial pool also runs in the differential // corpus; these are the named, committed subset.) // --------------------------------------------------------------------------- @@ -187,7 +182,7 @@ const FLOAT_POOL_VALUES: string[] = [ "6.097555160522461e-5", // max f16 subnormal "5.960464477539063e-8", // min f16 subnormal "5.960464477539064e-8", // next double up - NOT f16-exact → f64 - "2.9802322387695312e-8", // below min f16 subnormal → f32? (frozen behavior) + "2.9802322387695312e-8", // below min f16 subnormal // f32 precision cliff at 2^24. NOTE: whole values here integer-reduce in // cbor() DISPATCH (exact, no fround) - the Q1 fround collisions are only // reachable via bare Float nodes; see the float-simple section below. @@ -249,7 +244,7 @@ const floats: EncodeCorpusEntry[] = FLOAT_POOL_VALUES.map((v) => e(`float/${v}`, // 2b. Bare Cbor nodes - the attachMethods passthrough arm, the float // encoder's OWN reduction ladder (only reachable here: plain whole // numbers integer-reduce in dispatch before f64CborData ever runs), and -// the frozen quirks Q1/Q2 that are invisible through normal dispatch. +// the quirks Q1/Q2 that are invisible through normal dispatch. // --------------------------------------------------------------------------- const fsimple = (v: string): Recipe => ({ k: "floatsimple", v }); @@ -287,7 +282,7 @@ const bareNodes: EncodeCorpusEntry[] = [ e("rawnegmag/23-is-minus-24", { k: "rawnegmag", v: "23" }), // 0x37 e("rawnegmag/u64-max-is-minus-2^64", { k: "rawnegmag", v: "18446744073709551615" }), e("rawbad/malformed-bytestring-node-throws", { k: "rawbad" }), - // Unsupported input types - frozen Custom throws. + // Unsupported input types throw Custom. e("unsupported/symbol-throws", { k: "symbol" }), e("unsupported/function-throws", { k: "fn" }), ]; @@ -331,6 +326,15 @@ const strings: EncodeCorpusEntry[] = [ e("str/control-chars", s("\t\n\r")), e("str/lone-surrogate-becomes-replacement", s("\ud800")), // TextEncoder → U+FFFD e("str/448-byte-lorem-u16-head", sr("Lorem ipsum dolor sit amet, ", 16)), // 448 chars → 0x79 head + // Bare Text nodes bypass cbor(): NFC must still be applied by the ENCODER + // (Rust `cbor_data` parity), not only by the constructor. + e("rawtext/nfd-e-acute", { k: "rawtext", v: "é" }), // node holds 65cc81, encodes 62c3a9 + e("rawtext/nfd-in-array", arr({ k: "rawtext", v: "é" })), // 8162c3a9 + e("rawtext/nfc-passthrough", { k: "rawtext", v: "é" }), // already composed: 62c3a9 + e("rawtext/ascii-fast-path", { k: "rawtext", v: "plain ascii" }), + // Map keys are compared by encoded bytes: an NFD key and its NFC form are + // the SAME key, so the second insert replaces the first (a1 62c3a9 02). + e("rawtext/map-nfd-then-nfc-key", map([{ k: "rawtext", v: "é" }, n(1)], [s("é"), n(2)])), ]; // --------------------------------------------------------------------------- @@ -441,8 +445,8 @@ const maps: EncodeCorpusEntry[] = [ ]; // --------------------------------------------------------------------------- -// 8. JS Sets - INSERTION ORDER is preserved on the wire (frozen behavior, -// distinct from CborSet's canonical sort). +// 8. JS Sets - insertion order is preserved on the wire (distinct from +// CborSet's canonical sort). // --------------------------------------------------------------------------- const jsSets: EncodeCorpusEntry[] = [ @@ -517,7 +521,19 @@ const dates: EncodeCorpusEntry[] = [ e("date/y2038-plus", date(2147483648)), e("date/far-future", date(10000000000)), e("date/sub-ns-precision-dropped", date(1.0000000001)), - e("date/non-finite-throws", date("NaN")), + // `from_timestamp` parity: NaN saturates to the epoch (`trunc() as i64`), + // ±Infinity is rejected (the reference panics; TS throws InvalidDate). + e("date/nan-saturates-to-epoch", date("NaN")), // c100 + e("date/infinity-throws", date("Infinity")), + e("date/negative-infinity-throws", date("-Infinity")), + // The range check applies to the truncated whole seconds only, so a + // fraction below chrono's MIN still lands on MIN (executed on 0.25.2). + e("date/below-min-fraction-truncates", date("-8334601228800.5")), // c13b000007948cf211ff + e("date/below-min-fraction-999", date("-8334601228800.999")), + e("date/max-plus-fraction", date("8210266876799.5")), + e("date/max-plus-one-throws", date("8210266876800")), + e("date/min-minus-one-throws", date("-8334601228801")), + e("date/nanosecond-fraction-rounds-in-f64", date("1703500245.999999999")), // the f64 is …246 e("datestr/bare-date", datestr("2023-02-08")), e("datestr/rfc3339-utc", datestr("2023-02-08T15:30:45Z")), e("datestr/rfc3339-offset", datestr("2023-02-08T15:30:45+05:30")), @@ -598,16 +614,13 @@ const protocols: EncodeCorpusEntry[] = [ e("tocbor/int", tocbor(n(42))), e("tocbor/map", tocbor(map([n(1), n(2)]))), e("tocbor/in-array", arr(tocbor(s("x")), n(1))), - // P3.7 tombstone shape: object with taggedCbor() and no toCbor(). - e("taggedproto/simple", taggedproto(99, s("payload")), "P3.7"), - e("taggedproto/in-array", arr(taggedproto(99, n(1)), n(2)), "P3.7"), - e("taggedproto/as-map-value", map([s("k"), taggedproto(7, arr(n(1)))]), "P3.7"), - // Dispatch precedence: taggedCbor() wins over toCbor() today. The bytes - // reveal the winner (the toCbor side deliberately encodes differently). - // NOT tombstone-marked: post-P3.7 this shape doesn't throw - its bytes - // CHANGE (toCbor becomes the winner), and that flip lands as a reviewed - // fixture regeneration diff in the wave. - e("bothproto/taggedcbor-wins", { k: "bothproto", tag: "77", inner: s("x") }), + // Removed shape: object with taggedCbor() and no toCbor(). + e("taggedproto/simple", taggedproto(99, s("payload")), "tagged-cbor-only"), + e("taggedproto/in-array", arr(taggedproto(99, n(1)), n(2)), "tagged-cbor-only"), + e("taggedproto/as-map-value", map([s("k"), taggedproto(7, arr(n(1)))]), "tagged-cbor-only"), + // Dispatch precedence: toCbor() wins over taggedCbor(). The bytes reveal + // the winner (the toCbor side deliberately encodes differently). + e("bothproto/tocbor-wins", { k: "bothproto", tag: "77", inner: s("x") }), // Inherited tag/value (outer sniff trigger fires, own-keys check does not): // falls through to the plain-object→map branch, encoding only own entries. e("protoobj/inherited-tag-value-is-map", { @@ -626,12 +639,12 @@ const protocols: EncodeCorpusEntry[] = [ ], ownEntries: [], }), - // P3.5 tombstone shape: plain {tag, value} object literal. - e("tagobjlit/number-tag", tagobjlit(n(1), s("Hello")), "P3.5"), - e("tagobjlit/nested-value", tagobjlit(n(100), arr(n(1), n(2))), "P3.5"), - e("tagobjlit/string-tag-coerces", tagobjlit(s("24"), n(0)), "P3.5"), - e("tagobjlit/in-array", arr(tagobjlit(n(1), n(2))), "P3.5"), - e("tagobjlit/as-obj-value", obj(["inner", tagobjlit(n(5), n(6))]), "P3.5"), + // Removed shape: plain {tag, value} object literal. + e("tagobjlit/number-tag", tagobjlit(n(1), s("Hello")), "tag-value-literal"), + e("tagobjlit/nested-value", tagobjlit(n(100), arr(n(1), n(2))), "tag-value-literal"), + e("tagobjlit/string-tag-coerces", tagobjlit(s("24"), n(0)), "tag-value-literal"), + e("tagobjlit/in-array", arr(tagobjlit(n(1), n(2))), "tag-value-literal"), + e("tagobjlit/as-obj-value", obj(["inner", tagobjlit(n(5), n(6))]), "tag-value-literal"), ]; // --------------------------------------------------------------------------- diff --git a/tests/vectors/encode-vectors.json b/tests/vectors/encode-vectors.json index 4949095..d0915ea 100644 --- a/tests/vectors/encode-vectors.json +++ b/tests/vectors/encode-vectors.json @@ -1,7 +1,7 @@ { "//": "GENERATED by scripts/generate-vectors.ts - do not edit by hand. Review regenerated expectations before committing them.", - "sourceCommit": "351f16ffd6a0b13ab91b55a52ab03bbd7f53175c", - "count": 403, + "sourceCommit": "02766ae64e450096afd888380a8f80bcf24c2511", + "count": 416, "vectors": [ { "name": "int/zero", @@ -1639,8 +1639,7 @@ }, "expect": { "ok": true, - "hex": "fadf800000", - "decodeRejects": "NonCanonicalNumeric" + "hex": "fadf800000" } }, { @@ -1662,8 +1661,7 @@ }, "expect": { "ok": true, - "hex": "fa4f800000", - "decodeRejects": "NonCanonicalNumeric" + "hex": "fa4f800000" } }, { @@ -1674,8 +1672,7 @@ }, "expect": { "ok": true, - "hex": "fa4fc00000", - "decodeRejects": "NonCanonicalNumeric" + "hex": "fa4fc00000" } }, { @@ -1686,8 +1683,7 @@ }, "expect": { "ok": true, - "hex": "fa5a000000", - "decodeRejects": "NonCanonicalNumeric" + "hex": "fa5a000000" } }, { @@ -1698,8 +1694,7 @@ }, "expect": { "ok": true, - "hex": "fa5f000000", - "decodeRejects": "NonCanonicalNumeric" + "hex": "fa5f000000" } }, { @@ -2206,6 +2201,87 @@ "hex": "7901c04c6f72656d20697073756d20646f6c6f722073697420616d65742c204c6f72656d20697073756d20646f6c6f722073697420616d65742c204c6f72656d20697073756d20646f6c6f722073697420616d65742c204c6f72656d20697073756d20646f6c6f722073697420616d65742c204c6f72656d20697073756d20646f6c6f722073697420616d65742c204c6f72656d20697073756d20646f6c6f722073697420616d65742c204c6f72656d20697073756d20646f6c6f722073697420616d65742c204c6f72656d20697073756d20646f6c6f722073697420616d65742c204c6f72656d20697073756d20646f6c6f722073697420616d65742c204c6f72656d20697073756d20646f6c6f722073697420616d65742c204c6f72656d20697073756d20646f6c6f722073697420616d65742c204c6f72656d20697073756d20646f6c6f722073697420616d65742c204c6f72656d20697073756d20646f6c6f722073697420616d65742c204c6f72656d20697073756d20646f6c6f722073697420616d65742c204c6f72656d20697073756d20646f6c6f722073697420616d65742c204c6f72656d20697073756d20646f6c6f722073697420616d65742c20" } }, + { + "name": "rawtext/nfd-e-acute", + "recipe": { + "k": "rawtext", + "v": "é" + }, + "expect": { + "ok": true, + "hex": "62c3a9" + } + }, + { + "name": "rawtext/nfd-in-array", + "recipe": { + "k": "arr", + "items": [ + { + "k": "rawtext", + "v": "é" + } + ] + }, + "expect": { + "ok": true, + "hex": "8162c3a9" + } + }, + { + "name": "rawtext/nfc-passthrough", + "recipe": { + "k": "rawtext", + "v": "é" + }, + "expect": { + "ok": true, + "hex": "62c3a9" + } + }, + { + "name": "rawtext/ascii-fast-path", + "recipe": { + "k": "rawtext", + "v": "plain ascii" + }, + "expect": { + "ok": true, + "hex": "6b706c61696e206173636969" + } + }, + { + "name": "rawtext/map-nfd-then-nfc-key", + "recipe": { + "k": "map", + "entries": [ + [ + { + "k": "rawtext", + "v": "é" + }, + { + "k": "n", + "v": "1" + } + ], + [ + { + "k": "s", + "v": "é" + }, + { + "k": "n", + "v": "2" + } + ] + ] + }, + "expect": { + "ok": true, + "hex": "a162c3a902" + } + }, { "name": "bytes/empty", "recipe": { @@ -4523,16 +4599,104 @@ } }, { - "name": "date/non-finite-throws", + "name": "date/nan-saturates-to-epoch", "recipe": { "k": "date", "seconds": "NaN" }, + "expect": { + "ok": true, + "hex": "c100" + } + }, + { + "name": "date/infinity-throws", + "recipe": { + "k": "date", + "seconds": "Infinity" + }, + "expect": { + "ok": false, + "code": "InvalidDate" + } + }, + { + "name": "date/negative-infinity-throws", + "recipe": { + "k": "date", + "seconds": "-Infinity" + }, + "expect": { + "ok": false, + "code": "InvalidDate" + } + }, + { + "name": "date/below-min-fraction-truncates", + "recipe": { + "k": "date", + "seconds": "-8334601228800.5" + }, + "expect": { + "ok": true, + "hex": "c13b000007948cf211ff" + } + }, + { + "name": "date/below-min-fraction-999", + "recipe": { + "k": "date", + "seconds": "-8334601228800.999" + }, + "expect": { + "ok": true, + "hex": "c13b000007948cf211ff" + } + }, + { + "name": "date/max-plus-fraction", + "recipe": { + "k": "date", + "seconds": "8210266876799.5" + }, + "expect": { + "ok": true, + "hex": "c1fb429dde6829adfe00" + } + }, + { + "name": "date/max-plus-one-throws", + "recipe": { + "k": "date", + "seconds": "8210266876800" + }, "expect": { "ok": false, "code": "InvalidDate" } }, + { + "name": "date/min-minus-one-throws", + "recipe": { + "k": "date", + "seconds": "-8334601228801" + }, + "expect": { + "ok": false, + "code": "InvalidDate" + } + }, + { + "name": "date/nanosecond-fraction-rounds-in-f64", + "recipe": { + "k": "date", + "seconds": "1703500245.999999999" + }, + "expect": { + "ok": true, + "hex": "c11a658959d6" + } + }, { "name": "datestr/bare-date", "recipe": { @@ -5382,7 +5546,7 @@ "ok": false, "code": "Custom" }, - "tombstone": "P3.7" + "tombstone": "tagged-cbor-only" }, { "name": "taggedproto/in-array", @@ -5407,7 +5571,7 @@ "ok": false, "code": "Custom" }, - "tombstone": "P3.7" + "tombstone": "tagged-cbor-only" }, { "name": "taggedproto/as-map-value", @@ -5439,10 +5603,10 @@ "ok": false, "code": "Custom" }, - "tombstone": "P3.7" + "tombstone": "tagged-cbor-only" }, { - "name": "bothproto/taggedcbor-wins", + "name": "bothproto/tocbor-wins", "recipe": { "k": "bothproto", "tag": "77", @@ -5535,7 +5699,7 @@ "ok": false, "code": "Custom" }, - "tombstone": "P3.5" + "tombstone": "tag-value-literal" }, { "name": "tagobjlit/nested-value", @@ -5563,7 +5727,7 @@ "ok": false, "code": "Custom" }, - "tombstone": "P3.5" + "tombstone": "tag-value-literal" }, { "name": "tagobjlit/string-tag-coerces", @@ -5582,7 +5746,7 @@ "ok": false, "code": "Custom" }, - "tombstone": "P3.5" + "tombstone": "tag-value-literal" }, { "name": "tagobjlit/in-array", @@ -5606,7 +5770,7 @@ "ok": false, "code": "Custom" }, - "tombstone": "P3.5" + "tombstone": "tag-value-literal" }, { "name": "tagobjlit/as-obj-value", @@ -5633,7 +5797,7 @@ "ok": false, "code": "Custom" }, - "tombstone": "P3.5" + "tombstone": "tag-value-literal" }, { "name": "biguint/zero-empty-bstr", diff --git a/tests/vectors/format-corpus.ts b/tests/vectors/format-corpus.ts new file mode 100644 index 0000000..7f7f987 Binary files /dev/null and b/tests/vectors/format-corpus.ts differ diff --git a/tests/vectors/format-vectors.json b/tests/vectors/format-vectors.json new file mode 100644 index 0000000..20bb49f --- /dev/null +++ b/tests/vectors/format-vectors.json @@ -0,0 +1,2102 @@ +{ + "//": "GENERATED by scripts/generate-vectors.ts - do not edit by hand. Review regenerated expectations before committing them.", + "sourceCommit": "02766ae64e450096afd888380a8f80bcf24c2511", + "count": 102, + "vectors": [ + { + "name": "float/tie-f9000a", + "input": { + "hex": "f9000a" + }, + "config": "none", + "expect": { + "hex": "f9000a", + "diagnostic": "5.960464477539063e-7", + "annotated": "5.960464477539063e-7", + "flat": "5.960464477539063e-7", + "summary": "5.960464477539063e-7", + "hexAnnotated": "f9000a # 5.960464477539063e-7" + }, + "note": "10 * 2^-24: exact decimal tie, Rust rounds up" + }, + { + "name": "float/tie-f90032", + "input": { + "hex": "f90032" + }, + "config": "none", + "expect": { + "hex": "f90032", + "diagnostic": "2.9802322387695313e-6", + "annotated": "2.9802322387695313e-6", + "flat": "2.9802322387695313e-6", + "summary": "2.9802322387695313e-6", + "hexAnnotated": "f90032 # 2.9802322387695313e-6" + } + }, + { + "name": "float/tie-fa33000000", + "input": { + "hex": "fa33000000" + }, + "config": "none", + "expect": { + "hex": "fa33000000", + "diagnostic": "2.9802322387695313e-8", + "annotated": "2.9802322387695313e-8", + "flat": "2.9802322387695313e-8", + "summary": "2.9802322387695313e-8", + "hexAnnotated": "fa33000000 # 2.9802322387695313e-8" + } + }, + { + "name": "float/tie-fb4090000010000000", + "input": { + "hex": "fb4090000010000000" + }, + "config": "none", + "expect": { + "hex": "fb4090000010000000", + "diagnostic": "1024.0000610351563", + "annotated": "1024.0000610351563", + "flat": "1024.0000610351563", + "summary": "1024.0000610351563", + "hexAnnotated": "fb4090000010000000 # 1024.0000610351563" + } + }, + { + "name": "float/tie-fb4210000000000800", + "input": { + "hex": "fb4210000000000800" + }, + "config": "none", + "expect": { + "hex": "fb4210000000000800", + "diagnostic": "17179869184.007813", + "annotated": "17179869184.007813", + "flat": "17179869184.007813", + "summary": "17179869184.007813", + "hexAnnotated": "fb4210000000000800 # 17179869184.007813" + } + }, + { + "name": "float/1.5", + "input": { + "recipe": { + "k": "n", + "v": "1.5" + } + }, + "config": "none", + "expect": { + "hex": "f93e00", + "diagnostic": "1.5", + "annotated": "1.5", + "flat": "1.5", + "summary": "1.5", + "hexAnnotated": "f93e00 # 1.5" + } + }, + { + "name": "float/-1.5", + "input": { + "recipe": { + "k": "n", + "v": "-1.5" + } + }, + "config": "none", + "expect": { + "hex": "f9be00", + "diagnostic": "-1.5", + "annotated": "-1.5", + "flat": "-1.5", + "summary": "-1.5", + "hexAnnotated": "f9be00 # -1.5" + } + }, + { + "name": "float/pi", + "input": { + "recipe": { + "k": "n", + "v": "3.141592653589793" + } + }, + "config": "none", + "expect": { + "hex": "fb400921fb54442d18", + "diagnostic": "3.141592653589793", + "annotated": "3.141592653589793", + "flat": "3.141592653589793", + "summary": "3.141592653589793", + "hexAnnotated": "fb400921fb54442d18 # 3.141592653589793" + } + }, + { + "name": "float/1e21", + "input": { + "recipe": { + "k": "n", + "v": "1e21" + } + }, + "config": "none", + "expect": { + "hex": "fb444b1ae4d6e2ef50", + "diagnostic": "1e21", + "annotated": "1e21", + "flat": "1e21", + "summary": "1e21", + "hexAnnotated": "fb444b1ae4d6e2ef50 # 1e21" + } + }, + { + "name": "float/1.5e20", + "input": { + "recipe": { + "k": "n", + "v": "1.5e20" + } + }, + "config": "none", + "expect": { + "hex": "fb442043561a882930", + "diagnostic": "1.5e20", + "annotated": "1.5e20", + "flat": "1.5e20", + "summary": "1.5e20", + "hexAnnotated": "fb442043561a882930 # 1.5e20" + } + }, + { + "name": "float/-1.5e20", + "input": { + "recipe": { + "k": "n", + "v": "-1.5e20" + } + }, + "config": "none", + "expect": { + "hex": "fbc42043561a882930", + "diagnostic": "-1.5e20", + "annotated": "-1.5e20", + "flat": "-1.5e20", + "summary": "-1.5e20", + "hexAnnotated": "fbc42043561a882930 # -1.5e20" + } + }, + { + "name": "float/5e-324", + "input": { + "recipe": { + "k": "n", + "v": "5e-324" + } + }, + "config": "none", + "expect": { + "hex": "fb0000000000000001", + "diagnostic": "5e-324", + "annotated": "5e-324", + "flat": "5e-324", + "summary": "5e-324", + "hexAnnotated": "fb0000000000000001 # 5e-324" + } + }, + { + "name": "float/1e-5", + "input": { + "recipe": { + "k": "n", + "v": "0.00001" + } + }, + "config": "none", + "expect": { + "hex": "fb3ee4f8b588e368f1", + "diagnostic": "1e-5", + "annotated": "1e-5", + "flat": "1e-5", + "summary": "1e-5", + "hexAnnotated": "fb3ee4f8b588e368f1 # 1e-5" + } + }, + { + "name": "float/0.0001", + "input": { + "recipe": { + "k": "n", + "v": "0.0001" + } + }, + "config": "none", + "expect": { + "hex": "fb3f1a36e2eb1c432d", + "diagnostic": "0.0001", + "annotated": "0.0001", + "flat": "0.0001", + "summary": "0.0001", + "hexAnnotated": "fb3f1a36e2eb1c432d # 0.0001" + } + }, + { + "name": "float/0.1", + "input": { + "recipe": { + "k": "n", + "v": "0.1" + } + }, + "config": "none", + "expect": { + "hex": "fb3fb999999999999a", + "diagnostic": "0.1", + "annotated": "0.1", + "flat": "0.1", + "summary": "0.1", + "hexAnnotated": "fb3fb999999999999a # 0.1" + } + }, + { + "name": "float/123456789.125", + "input": { + "recipe": { + "k": "n", + "v": "123456789.125" + } + }, + "config": "none", + "expect": { + "hex": "fb419d6f3454800000", + "diagnostic": "123456789.125", + "annotated": "123456789.125", + "flat": "123456789.125", + "summary": "123456789.125", + "hexAnnotated": "fb419d6f3454800000 # 123456789.125" + } + }, + { + "name": "float/NaN", + "input": { + "recipe": { + "k": "n", + "v": "NaN" + } + }, + "config": "none", + "expect": { + "hex": "f97e00", + "diagnostic": "NaN", + "annotated": "NaN", + "flat": "NaN", + "summary": "NaN", + "hexAnnotated": "f97e00 # NaN" + } + }, + { + "name": "float/Infinity", + "input": { + "recipe": { + "k": "n", + "v": "Infinity" + } + }, + "config": "none", + "expect": { + "hex": "f97c00", + "diagnostic": "Infinity", + "annotated": "Infinity", + "flat": "Infinity", + "summary": "Infinity", + "hexAnnotated": "f97c00 # Infinity" + } + }, + { + "name": "float/-Infinity", + "input": { + "recipe": { + "k": "n", + "v": "-Infinity" + } + }, + "config": "none", + "expect": { + "hex": "f9fc00", + "diagnostic": "-Infinity", + "annotated": "-Infinity", + "flat": "-Infinity", + "summary": "-Infinity", + "hexAnnotated": "f9fc00 # -Infinity" + } + }, + { + "name": "float/f16-min-subnormal", + "input": { + "hex": "f90001" + }, + "config": "none", + "expect": { + "hex": "f90001", + "diagnostic": "5.960464477539063e-8", + "annotated": "5.960464477539063e-8", + "flat": "5.960464477539063e-8", + "summary": "5.960464477539063e-8", + "hexAnnotated": "f90001 # 5.960464477539063e-8" + } + }, + { + "name": "float/f32-max", + "input": { + "hex": "fa7f7fffff" + }, + "config": "none", + "expect": { + "hex": "fa7f7fffff", + "diagnostic": "3.4028234663852886e38", + "annotated": "3.4028234663852886e38", + "flat": "3.4028234663852886e38", + "summary": "3.4028234663852886e38", + "hexAnnotated": "fa7f7fffff # 3.4028234663852886e38" + } + }, + { + "name": "float/f64-max", + "input": { + "recipe": { + "k": "n", + "v": "1.7976931348623157e308" + } + }, + "config": "none", + "expect": { + "hex": "fb7fefffffffffffff", + "diagnostic": "1.7976931348623157e308", + "annotated": "1.7976931348623157e308", + "flat": "1.7976931348623157e308", + "summary": "1.7976931348623157e308", + "hexAnnotated": "fb7fefffffffffffff # 1.7976931348623157e308" + } + }, + { + "name": "float/whole-f32-head-stays-float", + "input": { + "hex": "fa4f800000" + }, + "config": "none", + "expect": { + "hex": "fa4f800000", + "diagnostic": "4294967296.0", + "annotated": "4294967296.0", + "flat": "4294967296.0", + "summary": "4294967296.0", + "hexAnnotated": "fa4f800000 # 4294967296.0" + }, + "note": "2^32 as f32 is accepted and stays a float node: prints 4294967296.0" + }, + { + "name": "float/whole-f32-head-becomes-integer", + "input": { + "hex": "fa4f000001" + }, + "config": "none", + "expect": { + "hex": "1a80000100", + "diagnostic": "2147483904", + "annotated": "2147483904", + "flat": "2147483904", + "summary": "2147483904", + "hexAnnotated": "1a80000100 # unsigned(2147483904)" + }, + "note": "2^31+256 as f32 decodes to the integer 2147483904" + }, + { + "name": "float/in-array", + "input": { + "recipe": { + "k": "arr", + "items": [ + { + "k": "n", + "v": "1.5" + }, + { + "k": "n", + "v": "NaN" + }, + { + "k": "n", + "v": "0" + }, + { + "k": "n", + "v": "1e21" + } + ] + } + }, + "config": "none", + "expect": { + "hex": "84f93e00f97e0000fb444b1ae4d6e2ef50", + "diagnostic": "[1.5, NaN, 0, 1e21]", + "annotated": "[1.5, NaN, 0, 1e21]", + "flat": "[1.5, NaN, 0, 1e21]", + "summary": "[1.5, NaN, 0, 1e21]", + "hexAnnotated": "84 # array(4)\n f93e00 # 1.5\n f97e00 # NaN\n 00 # unsigned(0)\n fb444b1ae4d6e2ef50 # 1e21" + } + }, + { + "name": "int/0", + "input": { + "recipe": { + "k": "n", + "v": "0" + } + }, + "config": "none", + "expect": { + "hex": "00", + "diagnostic": "0", + "annotated": "0", + "flat": "0", + "summary": "0", + "hexAnnotated": "00 # unsigned(0)" + } + }, + { + "name": "int/23", + "input": { + "recipe": { + "k": "n", + "v": "23" + } + }, + "config": "none", + "expect": { + "hex": "17", + "diagnostic": "23", + "annotated": "23", + "flat": "23", + "summary": "23", + "hexAnnotated": "17 # unsigned(23)" + } + }, + { + "name": "int/24", + "input": { + "recipe": { + "k": "n", + "v": "24" + } + }, + "config": "none", + "expect": { + "hex": "1818", + "diagnostic": "24", + "annotated": "24", + "flat": "24", + "summary": "24", + "hexAnnotated": "1818 # unsigned(24)" + } + }, + { + "name": "int/255", + "input": { + "recipe": { + "k": "n", + "v": "255" + } + }, + "config": "none", + "expect": { + "hex": "18ff", + "diagnostic": "255", + "annotated": "255", + "flat": "255", + "summary": "255", + "hexAnnotated": "18ff # unsigned(255)" + } + }, + { + "name": "int/256", + "input": { + "recipe": { + "k": "n", + "v": "256" + } + }, + "config": "none", + "expect": { + "hex": "190100", + "diagnostic": "256", + "annotated": "256", + "flat": "256", + "summary": "256", + "hexAnnotated": "190100 # unsigned(256)" + } + }, + { + "name": "int/65536", + "input": { + "recipe": { + "k": "n", + "v": "65536" + } + }, + "config": "none", + "expect": { + "hex": "1a00010000", + "diagnostic": "65536", + "annotated": "65536", + "flat": "65536", + "summary": "65536", + "hexAnnotated": "1a00010000 # unsigned(65536)" + } + }, + { + "name": "int/u32-max", + "input": { + "recipe": { + "k": "n", + "v": "4294967295" + } + }, + "config": "none", + "expect": { + "hex": "1affffffff", + "diagnostic": "4294967295", + "annotated": "4294967295", + "flat": "4294967295", + "summary": "4294967295", + "hexAnnotated": "1affffffff # unsigned(4294967295)" + } + }, + { + "name": "int/u64-max", + "input": { + "recipe": { + "k": "bi", + "v": "18446744073709551615" + } + }, + "config": "none", + "expect": { + "hex": "1bffffffffffffffff", + "diagnostic": "18446744073709551615", + "annotated": "18446744073709551615", + "flat": "18446744073709551615", + "summary": "18446744073709551615", + "hexAnnotated": "1bffffffffffffffff # unsigned(18446744073709551615)" + } + }, + { + "name": "int/2^53", + "input": { + "recipe": { + "k": "bi", + "v": "9007199254740992" + } + }, + "config": "none", + "expect": { + "hex": "1b0020000000000000", + "diagnostic": "9007199254740992", + "annotated": "9007199254740992", + "flat": "9007199254740992", + "summary": "9007199254740992", + "hexAnnotated": "1b0020000000000000 # unsigned(9007199254740992)" + } + }, + { + "name": "int/-1", + "input": { + "recipe": { + "k": "n", + "v": "-1" + } + }, + "config": "none", + "expect": { + "hex": "20", + "diagnostic": "-1", + "annotated": "-1", + "flat": "-1", + "summary": "-1", + "hexAnnotated": "20 # negative(-1)" + } + }, + { + "name": "int/-24", + "input": { + "recipe": { + "k": "n", + "v": "-24" + } + }, + "config": "none", + "expect": { + "hex": "37", + "diagnostic": "-24", + "annotated": "-24", + "flat": "-24", + "summary": "-24", + "hexAnnotated": "37 # negative(-24)" + } + }, + { + "name": "int/-25", + "input": { + "recipe": { + "k": "n", + "v": "-25" + } + }, + "config": "none", + "expect": { + "hex": "3818", + "diagnostic": "-25", + "annotated": "-25", + "flat": "-25", + "summary": "-25", + "hexAnnotated": "3818 # negative(-25)" + } + }, + { + "name": "int/i64-min", + "input": { + "recipe": { + "k": "bi", + "v": "-9223372036854775808" + } + }, + "config": "none", + "expect": { + "hex": "3b7fffffffffffffff", + "diagnostic": "-9223372036854775808", + "annotated": "-9223372036854775808", + "flat": "-9223372036854775808", + "summary": "-9223372036854775808", + "hexAnnotated": "3b7fffffffffffffff # negative(-9223372036854775808)" + } + }, + { + "name": "int/65-bit-negative", + "input": { + "hex": "3bffffffffffffffff" + }, + "config": "none", + "expect": { + "hex": "3bffffffffffffffff", + "diagnostic": "-18446744073709551616", + "annotated": "-18446744073709551616", + "flat": "-18446744073709551616", + "summary": "-18446744073709551616", + "hexAnnotated": "3bffffffffffffffff # negative(-18446744073709551616)" + }, + "note": "-2^64, not an i64" + }, + { + "name": "simple/true", + "input": { + "recipe": { + "k": "b", + "v": true + } + }, + "config": "none", + "expect": { + "hex": "f5", + "diagnostic": "true", + "annotated": "true", + "flat": "true", + "summary": "true", + "hexAnnotated": "f5 # true" + } + }, + { + "name": "simple/false", + "input": { + "recipe": { + "k": "b", + "v": false + } + }, + "config": "none", + "expect": { + "hex": "f4", + "diagnostic": "false", + "annotated": "false", + "flat": "false", + "summary": "false", + "hexAnnotated": "f4 # false" + } + }, + { + "name": "simple/null", + "input": { + "recipe": { + "k": "null" + } + }, + "config": "none", + "expect": { + "hex": "f6", + "diagnostic": "null", + "annotated": "null", + "flat": "null", + "summary": "null", + "hexAnnotated": "f6 # null" + } + }, + { + "name": "text/empty", + "input": { + "recipe": { + "k": "s", + "v": "" + } + }, + "config": "none", + "expect": { + "hex": "60", + "diagnostic": "\"\"", + "annotated": "\"\"", + "flat": "\"\"", + "summary": "\"\"", + "hexAnnotated": "60 # text(0)\n # \"\"" + } + }, + { + "name": "text/hello", + "input": { + "recipe": { + "k": "s", + "v": "Hello" + } + }, + "config": "none", + "expect": { + "hex": "6548656c6c6f", + "diagnostic": "\"Hello\"", + "annotated": "\"Hello\"", + "flat": "\"Hello\"", + "summary": "\"Hello\"", + "hexAnnotated": "65 # text(5)\n 48656c6c6f # \"Hello\"" + } + }, + { + "name": "text/escapes-quote-backslash", + "input": { + "recipe": { + "k": "s", + "v": "a\"b\\c" + } + }, + "config": "none", + "expect": { + "hex": "656122625c63", + "diagnostic": "\"a\\\"b\\c\"", + "annotated": "\"a\\\"b\\c\"", + "flat": "\"a\\\"b\\c\"", + "summary": "\"a\\\"b\\c\"", + "hexAnnotated": "65 # text(5)\n 6122625c63 # \"a\"b\\c\"" + } + }, + { + "name": "text/escapes-newline-tab-cr", + "input": { + "recipe": { + "k": "s", + "v": "line1\nline2\ttab\rcr" + } + }, + "config": "none", + "expect": { + "hex": "726c696e65310a6c696e6532097461620d6372", + "diagnostic": "\"line1\nline2\ttab\rcr\"", + "annotated": "\"line1\nline2\ttab\rcr\"", + "flat": "\"line1\nline2\ttab\rcr\"", + "summary": "\"line1\nline2\ttab\rcr\"", + "hexAnnotated": "72 # text(18)\n 6c696e65310a6c696e6532097461620d6372 # \"line1\nline2\ttab\rcr\"" + } + }, + { + "name": "text/control-chars", + "input": { + "recipe": { + "k": "s", + "v": "\u0000\u0001\u001f" + } + }, + "config": "none", + "expect": { + "hex": "6400011f7f", + "diagnostic": "\"\u0000\u0001\u001f\"", + "annotated": "\"\u0000\u0001\u001f\"", + "flat": "\"\u0000\u0001\u001f\"", + "summary": "\"\u0000\u0001\u001f\"", + "hexAnnotated": "64 # text(4)\n 00011f7f # \"\u0000\u0001\u001f\"" + } + }, + { + "name": "text/unicode-latin", + "input": { + "recipe": { + "k": "s", + "v": "café" + } + }, + "config": "none", + "expect": { + "hex": "65636166c3a9", + "diagnostic": "\"café\"", + "annotated": "\"café\"", + "flat": "\"café\"", + "summary": "\"café\"", + "hexAnnotated": "65 # text(5)\n 636166c3a9 # \"café\"" + } + }, + { + "name": "text/unicode-cjk", + "input": { + "recipe": { + "k": "s", + "v": "こんにちは" + } + }, + "config": "none", + "expect": { + "hex": "6fe38193e38293e381abe381a1e381af", + "diagnostic": "\"こんにちは\"", + "annotated": "\"こんにちは\"", + "flat": "\"こんにちは\"", + "summary": "\"こんにちは\"", + "hexAnnotated": "6f # text(15)\n e38193e38293e381abe381a1e381af # \"こんにちは\"" + } + }, + { + "name": "text/emoji", + "input": { + "recipe": { + "k": "s", + "v": "😀👍🏽" + } + }, + "config": "none", + "expect": { + "hex": "6cf09f9880f09f918df09f8fbd", + "diagnostic": "\"😀👍🏽\"", + "annotated": "\"😀👍🏽\"", + "flat": "\"😀👍🏽\"", + "summary": "\"😀👍🏽\"", + "hexAnnotated": "6c # text(12)\n f09f9880f09f918df09f8fbd # \"😀👍🏽\"" + } + }, + { + "name": "text/bom-prefixed", + "input": { + "hex": "64efbbbf61" + }, + "config": "none", + "expect": { + "hex": "64efbbbf61", + "diagnostic": "\"a\"", + "annotated": "\"a\"", + "flat": "\"a\"", + "summary": "\"a\"", + "hexAnnotated": "64 # text(4)\n efbbbf61 # \"a\"" + }, + "note": "U+FEFF is kept and printed" + }, + { + "name": "text/nfd-bare-node", + "input": { + "recipe": { + "k": "rawtext", + "v": "é" + } + }, + "config": "none", + "expect": { + "hex": "62c3a9", + "diagnostic": "\"é\"", + "annotated": "\"é\"", + "flat": "\"é\"", + "summary": "\"é\"", + "hexAnnotated": "63 # text(3)\n 65cc81 # \"é\"" + }, + "note": "the node holds NFD; diagnostic and hex dump show it, the wire is NFC" + }, + { + "name": "text/long-30", + "input": { + "recipe": { + "k": "s", + "v": "abcdefghijklmnopqrstuvwxyz0123" + } + }, + "config": "none", + "expect": { + "hex": "781e6162636465666768696a6b6c6d6e6f707172737475767778797a30313233", + "diagnostic": "\"abcdefghijklmnopqrstuvwxyz0123\"", + "annotated": "\"abcdefghijklmnopqrstuvwxyz0123\"", + "flat": "\"abcdefghijklmnopqrstuvwxyz0123\"", + "summary": "\"abcdefghijklmnopqrstuvwxyz0123\"", + "hexAnnotated": "78 1e # text(30)\n 6162636465666768696a6b6c6d6e6f707172737475767778797a30313233 # \"abcdefghijklmnopqrstuvwxyz0123\"" + } + }, + { + "name": "bytes/empty", + "input": { + "recipe": { + "k": "bytes", + "hex": "" + } + }, + "config": "none", + "expect": { + "hex": "40", + "diagnostic": "h''", + "annotated": "h''", + "flat": "h''", + "summary": "h''", + "hexAnnotated": "40 # bytes(0)" + } + }, + { + "name": "bytes/hello", + "input": { + "recipe": { + "k": "bytes", + "hex": "48656c6c6f" + } + }, + "config": "none", + "expect": { + "hex": "4548656c6c6f", + "diagnostic": "h'48656c6c6f'", + "annotated": "h'48656c6c6f'", + "flat": "h'48656c6c6f'", + "summary": "h'48656c6c6f'", + "hexAnnotated": "45 # bytes(5)\n 48656c6c6f # \"Hello\"" + } + }, + { + "name": "bytes/astral", + "input": { + "recipe": { + "k": "bytes", + "hex": "f09f9880" + } + }, + "config": "none", + "expect": { + "hex": "44f09f9880", + "diagnostic": "h'f09f9880'", + "annotated": "h'f09f9880'", + "flat": "h'f09f9880'", + "summary": "h'f09f9880'", + "hexAnnotated": "44 # bytes(4)\n f09f9880 # \"😀\"" + } + }, + { + "name": "bytes/nul-then-astral", + "input": { + "recipe": { + "k": "bytes", + "hex": "00f09f9880" + } + }, + "config": "none", + "expect": { + "hex": "4500f09f9880", + "diagnostic": "h'00f09f9880'", + "annotated": "h'00f09f9880'", + "flat": "h'00f09f9880'", + "summary": "h'00f09f9880'", + "hexAnnotated": "45 # bytes(5)\n 00f09f9880 # \".😀\"" + } + }, + { + "name": "bytes/astral-and-bmp", + "input": { + "recipe": { + "k": "bytes", + "hex": "f0908080e29c93" + } + }, + "config": "none", + "expect": { + "hex": "47f0908080e29c93", + "diagnostic": "h'f0908080e29c93'", + "annotated": "h'f0908080e29c93'", + "flat": "h'f0908080e29c93'", + "summary": "h'f0908080e29c93'", + "hexAnnotated": "47 # bytes(7)\n f0908080e29c93 # \"𐀀✓\"" + } + }, + { + "name": "bytes/control-then-ascii", + "input": { + "recipe": { + "k": "bytes", + "hex": "0a41" + } + }, + "config": "none", + "expect": { + "hex": "420a41", + "diagnostic": "h'0a41'", + "annotated": "h'0a41'", + "flat": "h'0a41'", + "summary": "h'0a41'", + "hexAnnotated": "42 # bytes(2)\n 0a41 # \".A\"" + } + }, + { + "name": "bytes/single-nul", + "input": { + "recipe": { + "k": "bytes", + "hex": "00" + } + }, + "config": "none", + "expect": { + "hex": "4100", + "diagnostic": "h'00'", + "annotated": "h'00'", + "flat": "h'00'", + "summary": "h'00'", + "hexAnnotated": "41 # bytes(1)\n 00" + } + }, + { + "name": "bytes/bom-then-ascii", + "input": { + "recipe": { + "k": "bytes", + "hex": "efbbbf41" + } + }, + "config": "none", + "expect": { + "hex": "44efbbbf41", + "diagnostic": "h'efbbbf41'", + "annotated": "h'efbbbf41'", + "flat": "h'efbbbf41'", + "summary": "h'efbbbf41'", + "hexAnnotated": "44 # bytes(4)\n efbbbf41 # \"A\"" + } + }, + { + "name": "bytes/invalid-utf8", + "input": { + "recipe": { + "k": "bytes", + "hex": "ff00ff" + } + }, + "config": "none", + "expect": { + "hex": "43ff00ff", + "diagnostic": "h'ff00ff'", + "annotated": "h'ff00ff'", + "flat": "h'ff00ff'", + "summary": "h'ff00ff'", + "hexAnnotated": "43 # bytes(3)\n ff00ff" + } + }, + { + "name": "bytes/lorem-41", + "input": { + "recipe": { + "k": "bytes", + "hex": "536f6d65206d7973746572696573206172656e2774206d65616e7420746f20626520736f6c7665642e" + } + }, + "config": "none", + "expect": { + "hex": "5829536f6d65206d7973746572696573206172656e2774206d65616e7420746f20626520736f6c7665642e", + "diagnostic": "h'536f6d65206d7973746572696573206172656e2774206d65616e7420746f20626520736f6c7665642e'", + "annotated": "h'536f6d65206d7973746572696573206172656e2774206d65616e7420746f20626520736f6c7665642e'", + "flat": "h'536f6d65206d7973746572696573206172656e2774206d65616e7420746f20626520736f6c7665642e'", + "summary": "h'536f6d65206d7973746572696573206172656e2774206d65616e7420746f20626520736f6c7665642e'", + "hexAnnotated": "5829 # bytes(41)\n 536f6d65206d7973746572696573206172656e2774206d65616e7420746f20626520736f6c7665642e # \"Some mysteries aren't meant to be solved.\"" + } + }, + { + "name": "bytes/64-cycling", + "input": { + "recipe": { + "k": "br", + "start": 0, + "count": 64 + } + }, + "config": "none", + "expect": { + "hex": "5840000102030405060708090a0b0c0d0e0f101112131415161718191a1b1c1d1e1f202122232425262728292a2b2c2d2e2f303132333435363738393a3b3c3d3e3f", + "diagnostic": "h'000102030405060708090a0b0c0d0e0f101112131415161718191a1b1c1d1e1f202122232425262728292a2b2c2d2e2f303132333435363738393a3b3c3d3e3f'", + "annotated": "h'000102030405060708090a0b0c0d0e0f101112131415161718191a1b1c1d1e1f202122232425262728292a2b2c2d2e2f303132333435363738393a3b3c3d3e3f'", + "flat": "h'000102030405060708090a0b0c0d0e0f101112131415161718191a1b1c1d1e1f202122232425262728292a2b2c2d2e2f303132333435363738393a3b3c3d3e3f'", + "summary": "h'000102030405060708090a0b0c0d0e0f101112131415161718191a1b1c1d1e1f202122232425262728292a2b2c2d2e2f303132333435363738393a3b3c3d3e3f'", + "hexAnnotated": "5840 # bytes(64)\n 000102030405060708090a0b0c0d0e0f101112131415161718191a1b1c1d1e1f202122232425262728292a2b2c2d2e2f303132333435363738393a3b3c3d3e3f # \"................................ !\"#$%&'()*+,-./0123456789:;<=>?\"" + } + }, + { + "name": "array/empty", + "input": { + "recipe": { + "k": "arr", + "items": [] + } + }, + "config": "none", + "expect": { + "hex": "80", + "diagnostic": "[]", + "annotated": "[]", + "flat": "[]", + "summary": "[]", + "hexAnnotated": "80 # array(0)" + } + }, + { + "name": "array/ints", + "input": { + "recipe": { + "k": "arr", + "items": [ + { + "k": "n", + "v": "1" + }, + { + "k": "n", + "v": "2" + }, + { + "k": "n", + "v": "3" + } + ] + } + }, + "config": "none", + "expect": { + "hex": "83010203", + "diagnostic": "[1, 2, 3]", + "annotated": "[1, 2, 3]", + "flat": "[1, 2, 3]", + "summary": "[1, 2, 3]", + "hexAnnotated": "83 # array(3)\n 01 # unsigned(1)\n 02 # unsigned(2)\n 03 # unsigned(3)" + } + }, + { + "name": "array/nested", + "input": { + "recipe": { + "k": "arr", + "items": [ + { + "k": "arr", + "items": [ + { + "k": "n", + "v": "1" + }, + { + "k": "arr", + "items": [ + { + "k": "n", + "v": "2" + }, + { + "k": "arr", + "items": [ + { + "k": "n", + "v": "3" + } + ] + } + ] + } + ] + } + ] + } + }, + "config": "none", + "expect": { + "hex": "81820182028103", + "diagnostic": "[\n [\n 1,\n [\n 2,\n [3]\n ]\n ]\n]", + "annotated": "[\n [\n 1,\n [\n 2,\n [3]\n ]\n ]\n]", + "flat": "[[1, [2, [3]]]]", + "summary": "[[1, [2, [3]]]]", + "hexAnnotated": "81 # array(1)\n 82 # array(2)\n 01 # unsigned(1)\n 82 # array(2)\n 02 # unsigned(2)\n 81 # array(1)\n 03 # unsigned(3)" + } + }, + { + "name": "array/mixed", + "input": { + "recipe": { + "k": "arr", + "items": [ + { + "k": "s", + "v": "a" + }, + { + "k": "n", + "v": "1" + }, + { + "k": "b", + "v": true + }, + { + "k": "null" + }, + { + "k": "n", + "v": "1.5" + }, + { + "k": "bytes", + "hex": "0102" + } + ] + } + }, + "config": "none", + "expect": { + "hex": "86616101f5f6f93e00420102", + "diagnostic": "[\n \"a\",\n 1,\n true,\n null,\n 1.5,\n h'0102'\n]", + "annotated": "[\n \"a\",\n 1,\n true,\n null,\n 1.5,\n h'0102'\n]", + "flat": "[\"a\", 1, true, null, 1.5, h'0102']", + "summary": "[\"a\", 1, true, null, 1.5, h'0102']", + "hexAnnotated": "86 # array(6)\n 61 # text(1)\n 61 # \"a\"\n 01 # unsigned(1)\n f5 # true\n f6 # null\n f93e00 # 1.5\n 42 # bytes(2)\n 0102" + } + }, + { + "name": "array/strings-over-20-bytes", + "input": { + "recipe": { + "k": "arr", + "items": [ + { + "k": "s", + "v": "unicode ✓ ☺ 日本" + } + ] + } + }, + "config": "none", + "expect": { + "hex": "8176756e69636f646520e29c9320e298ba20e697a5e69cac", + "diagnostic": "[\n \"unicode ✓ ☺ 日本\"\n]", + "annotated": "[\n \"unicode ✓ ☺ 日本\"\n]", + "flat": "[\"unicode ✓ ☺ 日本\"]", + "summary": "[\"unicode ✓ ☺ 日本\"]", + "hexAnnotated": "81 # array(1)\n 76 # text(22)\n 756e69636f646520e29c9320e298ba20e697a5e69cac # \"unicode ✓ ☺ 日本\"" + } + }, + { + "name": "array/strings-under-20-bytes", + "input": { + "recipe": { + "k": "arr", + "items": [ + { + "k": "s", + "v": "eighteen chars!!!!" + } + ] + } + }, + "config": "none", + "expect": { + "hex": "8172656967687465656e20636861727321212121", + "diagnostic": "[\"eighteen chars!!!!\"]", + "annotated": "[\"eighteen chars!!!!\"]", + "flat": "[\"eighteen chars!!!!\"]", + "summary": "[\"eighteen chars!!!!\"]", + "hexAnnotated": "81 # array(1)\n 72 # text(18)\n 656967687465656e20636861727321212121 # \"eighteen chars!!!!\"" + } + }, + { + "name": "array/two-long-strings", + "input": { + "recipe": { + "k": "arr", + "items": [ + { + "k": "s", + "v": "✓✓✓✓" + }, + { + "k": "s", + "v": "☺☺☺☺" + } + ] + } + }, + "config": "none", + "expect": { + "hex": "826ce29c93e29c93e29c93e29c936ce298bae298bae298bae298ba", + "diagnostic": "[\n \"✓✓✓✓\",\n \"☺☺☺☺\"\n]", + "annotated": "[\n \"✓✓✓✓\",\n \"☺☺☺☺\"\n]", + "flat": "[\"✓✓✓✓\", \"☺☺☺☺\"]", + "summary": "[\"✓✓✓✓\", \"☺☺☺☺\"]", + "hexAnnotated": "82 # array(2)\n 6c # text(12)\n e29c93e29c93e29c93e29c93 # \"✓✓✓✓\"\n 6c # text(12)\n e298bae298bae298bae298ba # \"☺☺☺☺\"" + } + }, + { + "name": "map/empty", + "input": { + "recipe": { + "k": "map", + "entries": [] + } + }, + "config": "none", + "expect": { + "hex": "a0", + "diagnostic": "{}", + "annotated": "{}", + "flat": "{}", + "summary": "{}", + "hexAnnotated": "a0 # map(0)" + } + }, + { + "name": "map/one", + "input": { + "recipe": { + "k": "map", + "entries": [ + [ + { + "k": "n", + "v": "1" + }, + { + "k": "n", + "v": "2" + } + ] + ] + } + }, + "config": "none", + "expect": { + "hex": "a10102", + "diagnostic": "{1: 2}", + "annotated": "{1: 2}", + "flat": "{1: 2}", + "summary": "{1: 2}", + "hexAnnotated": "a1 # map(1)\n 01 # unsigned(1)\n 02 # unsigned(2)" + } + }, + { + "name": "map/text-keys", + "input": { + "recipe": { + "k": "map", + "entries": [ + [ + { + "k": "s", + "v": "a" + }, + { + "k": "n", + "v": "1" + } + ], + [ + { + "k": "s", + "v": "b" + }, + { + "k": "arr", + "items": [ + { + "k": "n", + "v": "1" + }, + { + "k": "n", + "v": "2" + } + ] + } + ] + ] + } + }, + "config": "none", + "expect": { + "hex": "a26161016162820102", + "diagnostic": "{\n \"a\":\n 1,\n \"b\":\n [1, 2]\n}", + "annotated": "{\n \"a\":\n 1,\n \"b\":\n [1, 2]\n}", + "flat": "{\"a\": 1, \"b\": [1, 2]}", + "summary": "{\"a\": 1, \"b\": [1, 2]}", + "hexAnnotated": "a2 # map(2)\n 61 # text(1)\n 61 # \"a\"\n 01 # unsigned(1)\n 61 # text(1)\n 62 # \"b\"\n 82 # array(2)\n 01 # unsigned(1)\n 02 # unsigned(2)" + } + }, + { + "name": "map/mixed-keys", + "input": { + "recipe": { + "k": "map", + "entries": [ + [ + { + "k": "n", + "v": "1" + }, + { + "k": "s", + "v": "int" + } + ], + [ + { + "k": "s", + "v": "k" + }, + { + "k": "s", + "v": "text" + } + ], + [ + { + "k": "bytes", + "hex": "01" + }, + { + "k": "s", + "v": "bytes" + } + ], + [ + { + "k": "arr", + "items": [ + { + "k": "n", + "v": "1" + } + ] + }, + { + "k": "s", + "v": "array" + } + ], + [ + { + "k": "n", + "v": "-1" + }, + { + "k": "s", + "v": "neg" + } + ] + ] + } + }, + "config": "none", + "expect": { + "hex": "a50163696e7420636e65674101656279746573616b64746578748101656172726179", + "diagnostic": "{\n 1:\n \"int\",\n -1:\n \"neg\",\n h'01':\n \"bytes\",\n \"k\":\n \"text\",\n [1]:\n \"array\"\n}", + "annotated": "{\n 1:\n \"int\",\n -1:\n \"neg\",\n h'01':\n \"bytes\",\n \"k\":\n \"text\",\n [1]:\n \"array\"\n}", + "flat": "{1: \"int\", -1: \"neg\", h'01': \"bytes\", \"k\": \"text\", [1]: \"array\"}", + "summary": "{1: \"int\", -1: \"neg\", h'01': \"bytes\", \"k\": \"text\", [1]: \"array\"}", + "hexAnnotated": "a5 # map(5)\n 01 # unsigned(1)\n 63 # text(3)\n 696e74 # \"int\"\n 20 # negative(-1)\n 63 # text(3)\n 6e6567 # \"neg\"\n 41 # bytes(1)\n 01\n 65 # text(5)\n 6279746573 # \"bytes\"\n 61 # text(1)\n 6b # \"k\"\n 64 # text(4)\n 74657874 # \"text\"\n 81 # array(1)\n 01 # unsigned(1)\n 65 # text(5)\n 6172726179 # \"array\"" + } + }, + { + "name": "map/nested-in-array", + "input": { + "recipe": { + "k": "arr", + "items": [ + { + "k": "map", + "entries": [ + [ + { + "k": "s", + "v": "k" + }, + { + "k": "s", + "v": "unicode ✓ ☺ 日本" + } + ] + ] + }, + { + "k": "n", + "v": "1" + } + ] + } + }, + "config": "none", + "expect": { + "hex": "82a1616b76756e69636f646520e29c9320e298ba20e697a5e69cac01", + "diagnostic": "[\n {\n \"k\":\n \"unicode ✓ ☺ 日本\"\n },\n 1\n]", + "annotated": "[\n {\n \"k\":\n \"unicode ✓ ☺ 日本\"\n },\n 1\n]", + "flat": "[{\"k\": \"unicode ✓ ☺ 日本\"}, 1]", + "summary": "[{\"k\": \"unicode ✓ ☺ 日本\"}, 1]", + "hexAnnotated": "82 # array(2)\n a1 # map(1)\n 61 # text(1)\n 6b # \"k\"\n 76 # text(22)\n 756e69636f646520e29c9320e298ba20e697a5e69cac # \"unicode ✓ ☺ 日本\"\n 01 # unsigned(1)" + } + }, + { + "name": "map/long-values", + "input": { + "recipe": { + "k": "map", + "entries": [ + [ + { + "k": "s", + "v": "k" + }, + { + "k": "s", + "v": "unicode ✓ ☺ 日本" + } + ] + ] + } + }, + "config": "none", + "expect": { + "hex": "a1616b76756e69636f646520e29c9320e298ba20e697a5e69cac", + "diagnostic": "{\n \"k\":\n \"unicode ✓ ☺ 日本\"\n}", + "annotated": "{\n \"k\":\n \"unicode ✓ ☺ 日本\"\n}", + "flat": "{\"k\": \"unicode ✓ ☺ 日本\"}", + "summary": "{\"k\": \"unicode ✓ ☺ 日本\"}", + "hexAnnotated": "a1 # map(1)\n 61 # text(1)\n 6b # \"k\"\n 76 # text(22)\n 756e69636f646520e29c9320e298ba20e697a5e69cac # \"unicode ✓ ☺ 日本\"" + } + }, + { + "name": "map/float-key", + "input": { + "recipe": { + "k": "map", + "entries": [ + [ + { + "k": "n", + "v": "1.5" + }, + { + "k": "s", + "v": "f" + } + ] + ] + } + }, + "config": "none", + "expect": { + "hex": "a1f93e006166", + "diagnostic": "{1.5: \"f\"}", + "annotated": "{1.5: \"f\"}", + "flat": "{1.5: \"f\"}", + "summary": "{1.5: \"f\"}", + "hexAnnotated": "a1 # map(1)\n f93e00 # 1.5\n 61 # text(1)\n 66 # \"f\"" + } + }, + { + "name": "tag/date-int-none", + "input": { + "recipe": { + "k": "tagged", + "tag": "1", + "content": { + "k": "n", + "v": "1675854714" + } + } + }, + "config": "none", + "expect": { + "hex": "c11a63e3837a", + "diagnostic": "1(1675854714)", + "annotated": "1(1675854714)", + "flat": "1(1675854714)", + "summary": "1(1675854714)", + "hexAnnotated": "c1 # tag(1)\n 1a63e3837a # unsigned(1675854714)" + } + }, + { + "name": "tag/date-int-standard", + "input": { + "recipe": { + "k": "tagged", + "tag": "1", + "content": { + "k": "n", + "v": "1675854714" + } + } + }, + "config": "standard", + "expect": { + "hex": "c11a63e3837a", + "diagnostic": "1(1675854714)", + "annotated": "1(1675854714) / date /", + "flat": "1(1675854714)", + "summary": "2023-02-08T11:11:54Z", + "hexAnnotated": "c1 # tag(1) date\n 1a63e3837a # unsigned(1675854714)" + } + }, + { + "name": "tag/date-float-standard", + "input": { + "recipe": { + "k": "tagged", + "tag": "1", + "content": { + "k": "n", + "v": "1675854714.5" + } + } + }, + "config": "standard", + "expect": { + "hex": "c1fb41d8f8e0dea00000", + "diagnostic": "1(1675854714.5)", + "annotated": "1(1675854714.5) / date /", + "flat": "1(1675854714.5)", + "summary": "2023-02-08T11:11:54Z", + "hexAnnotated": "c1 # tag(1) date\n fb41d8f8e0dea00000 # 1675854714.5" + } + }, + { + "name": "tag/date-negative-standard", + "input": { + "recipe": { + "k": "tagged", + "tag": "1", + "content": { + "k": "n", + "v": "-1" + } + } + }, + "config": "standard", + "expect": { + "hex": "c120", + "diagnostic": "1(-1)", + "annotated": "1(-1) / date /", + "flat": "1(-1)", + "summary": "1969-12-31T23:59:59Z", + "hexAnnotated": "c1 # tag(1) date\n 20 # negative(-1)" + } + }, + { + "name": "tag/date-nan-standard", + "input": { + "hex": "c1f97e00" + }, + "config": "standard", + "expect": { + "hex": "c1f97e00", + "diagnostic": "1(NaN)", + "annotated": "1(NaN) / date /", + "flat": "1(NaN)", + "summary": "1970-01-01", + "hexAnnotated": "c1 # tag(1) date\n f97e00 # NaN" + }, + "note": "NaN saturates to the epoch in the summarizer" + }, + { + "name": "tag/date-text-content-standard", + "input": { + "recipe": { + "k": "tagged", + "tag": "1", + "content": { + "k": "s", + "v": "x" + } + } + }, + "config": "standard", + "expect": { + "hex": "c16178", + "diagnostic": "1(\"x\")", + "annotated": "1(\"x\") / date /", + "flat": "1(\"x\")", + "summary": "", + "hexAnnotated": "c1 # tag(1) date\n 61 # text(1)\n 78 # \"x\"" + }, + "note": "summarizer error" + }, + { + "name": "tag/date-recipe-standard", + "input": { + "recipe": { + "k": "date", + "seconds": "0.5" + } + }, + "config": "standard", + "expect": { + "hex": "c1f93800", + "diagnostic": "1(0.5)", + "annotated": "1(0.5) / date /", + "flat": "1(0.5)", + "summary": "1970-01-01", + "hexAnnotated": "c1 # tag(1) date\n f93800 # 0.5" + } + }, + { + "name": "tag/date-string-standard", + "input": { + "recipe": { + "k": "datestr", + "v": "2023-02-08T15:30:45Z" + } + }, + "config": "standard", + "expect": { + "hex": "c11a63e3c025", + "diagnostic": "1(1675870245)", + "annotated": "1(1675870245) / date /", + "flat": "1(1675870245)", + "summary": "2023-02-08T15:30:45Z", + "hexAnnotated": "c1 # tag(1) date\n 1a63e3c025 # unsigned(1675870245)" + } + }, + { + "name": "tag/date-leap-second-standard", + "input": { + "recipe": { + "k": "datestr", + "v": "2023-12-25T10:30:60Z" + } + }, + "config": "standard", + "expect": { + "hex": "c11a658959e4", + "diagnostic": "1(1703500260)", + "annotated": "1(1703500260) / date /", + "flat": "1(1703500260)", + "summary": "2023-12-25T10:31:00Z", + "hexAnnotated": "c1 # tag(1) date\n 1a658959e4 # unsigned(1703500260)" + }, + "note": "the wire value is a plain instant: summary prints 10:31:00Z" + }, + { + "name": "tag/date-min-standard", + "input": { + "hex": "c13b000007948cf211ff" + }, + "config": "standard", + "expect": { + "hex": "c13b000007948cf211ff", + "diagnostic": "1(-8334601228800)", + "annotated": "1(-8334601228800) / date /", + "flat": "1(-8334601228800)", + "summary": "-262143-01-01", + "hexAnnotated": "c1 # tag(1) date\n 3b000007948cf211ff # negative(-8334601228800)" + } + }, + { + "name": "tag/tag-of-tag", + "input": { + "hex": "c1c100" + }, + "config": "standard", + "expect": { + "hex": "c1c100", + "diagnostic": "1(\n 1(0)\n)", + "annotated": "1( / date /\n 1(0) / date /\n)", + "flat": "1(1(0))", + "summary": "", + "hexAnnotated": "c1 # tag(1) date\n c1 # tag(1) date\n 00 # unsigned(0)" + } + }, + { + "name": "tag/custom-unnamed", + "input": { + "recipe": { + "k": "tagged", + "tag": "40000", + "content": { + "k": "arr", + "items": [ + { + "k": "n", + "v": "1" + }, + { + "k": "n", + "v": "2" + }, + { + "k": "n", + "v": "3" + } + ] + } + } + }, + "config": "none", + "expect": { + "hex": "d99c4083010203", + "diagnostic": "40000(\n [1, 2, 3]\n)", + "annotated": "40000(\n [1, 2, 3]\n)", + "flat": "40000([1, 2, 3])", + "summary": "40000([1, 2, 3])", + "hexAnnotated": "d9 9c40 # tag(40000)\n 83 # array(3)\n 01 # unsigned(1)\n 02 # unsigned(2)\n 03 # unsigned(3)" + } + }, + { + "name": "tag/custom-unnamed-standard", + "input": { + "recipe": { + "k": "tagged", + "tag": "40000", + "content": { + "k": "s", + "v": "x" + } + } + }, + "config": "standard", + "expect": { + "hex": "d99c406178", + "diagnostic": "40000(\"x\")", + "annotated": "40000(\"x\")", + "flat": "40000(\"x\")", + "summary": "40000(\"x\")", + "hexAnnotated": "d9 9c40 # tag(40000)\n 61 # text(1)\n 78 # \"x\"" + } + }, + { + "name": "tag/nested-in-map", + "input": { + "recipe": { + "k": "map", + "entries": [ + [ + { + "k": "s", + "v": "d" + }, + { + "k": "tagged", + "tag": "1", + "content": { + "k": "n", + "v": "0" + } + } + ] + ] + } + }, + "config": "standard", + "expect": { + "hex": "a16164c100", + "diagnostic": "{\n \"d\":\n 1(0)\n}", + "annotated": "{\n \"d\":\n 1(0) / date /\n}", + "flat": "{\"d\": 1(0)}", + "summary": "{\"d\": 1970-01-01}", + "hexAnnotated": "a1 # map(1)\n 61 # text(1)\n 64 # \"d\"\n c1 # tag(1) date\n 00 # unsigned(0)" + } + }, + { + "name": "tag/large-tag-value", + "input": { + "hex": "dbffffffffffffffff00" + }, + "config": "none", + "expect": { + "hex": "dbffffffffffffffff00", + "diagnostic": "18446744073709551615(0)", + "annotated": "18446744073709551615(0)", + "flat": "18446744073709551615(0)", + "summary": "18446744073709551615(0)", + "hexAnnotated": "db ffffffffffffffff # tag(18446744073709551615)\n 00 # unsigned(0)" + } + }, + { + "name": "tag/bignum-2-none", + "input": { + "recipe": { + "k": "biguint", + "v": "18446744073709551616" + } + }, + "config": "none", + "expect": { + "hex": "c249010000000000000000", + "diagnostic": "2(\n h'010000000000000000'\n)", + "annotated": "2(\n h'010000000000000000'\n)", + "flat": "2(h'010000000000000000')", + "summary": "2(h'010000000000000000')", + "hexAnnotated": "c2 # tag(2)\n 49 # bytes(9)\n 010000000000000000" + } + }, + { + "name": "tag/bignum-2-standard-default-build", + "input": { + "recipe": { + "k": "biguint", + "v": "18446744073709551616" + } + }, + "config": "standard", + "build": "default", + "expect": { + "hex": "c249010000000000000000", + "diagnostic": "2(\n h'010000000000000000'\n)", + "annotated": "2(\n h'010000000000000000'\n)", + "flat": "2(h'010000000000000000')", + "summary": "2(h'010000000000000000')", + "hexAnnotated": "c2 # tag(2)\n 49 # bytes(9)\n 010000000000000000" + } + }, + { + "name": "tag/bignum-2-standard-bignum-build", + "input": { + "recipe": { + "k": "biguint", + "v": "18446744073709551616" + } + }, + "config": "standard+bignum", + "build": "bignum", + "expect": { + "hex": "c249010000000000000000", + "diagnostic": "2(\n h'010000000000000000'\n)", + "annotated": "2( / positive-bignum /\n h'010000000000000000'\n)", + "flat": "2(h'010000000000000000')", + "summary": "bignum(18446744073709551616)", + "hexAnnotated": "c2 # tag(2) positive-bignum\n 49 # bytes(9)\n 010000000000000000" + } + }, + { + "name": "tag/bignum-3-none", + "input": { + "recipe": { + "k": "bignum", + "v": "-18446744073709551617" + } + }, + "config": "none", + "expect": { + "hex": "c349010000000000000000", + "diagnostic": "3(\n h'010000000000000000'\n)", + "annotated": "3(\n h'010000000000000000'\n)", + "flat": "3(h'010000000000000000')", + "summary": "3(h'010000000000000000')", + "hexAnnotated": "c3 # tag(3)\n 49 # bytes(9)\n 010000000000000000" + } + }, + { + "name": "tag/bignum-3-standard-default-build", + "input": { + "recipe": { + "k": "bignum", + "v": "-18446744073709551617" + } + }, + "config": "standard", + "build": "default", + "expect": { + "hex": "c349010000000000000000", + "diagnostic": "3(\n h'010000000000000000'\n)", + "annotated": "3(\n h'010000000000000000'\n)", + "flat": "3(h'010000000000000000')", + "summary": "3(h'010000000000000000')", + "hexAnnotated": "c3 # tag(3)\n 49 # bytes(9)\n 010000000000000000" + } + }, + { + "name": "tag/bignum-3-standard-bignum-build", + "input": { + "recipe": { + "k": "bignum", + "v": "-18446744073709551617" + } + }, + "config": "standard+bignum", + "build": "bignum", + "expect": { + "hex": "c349010000000000000000", + "diagnostic": "3(\n h'010000000000000000'\n)", + "annotated": "3( / negative-bignum /\n h'010000000000000000'\n)", + "flat": "3(h'010000000000000000')", + "summary": "bignum(-18446744073709551617)", + "hexAnnotated": "c3 # tag(3) negative-bignum\n 49 # bytes(9)\n 010000000000000000" + } + }, + { + "name": "tag/bignum-2-empty-bytes-bignum-build", + "input": { + "hex": "c240" + }, + "config": "standard+bignum", + "build": "bignum", + "expect": { + "hex": "c240", + "diagnostic": "2(h'')", + "annotated": "2(h'') / positive-bignum /", + "flat": "2(h'')", + "summary": "bignum(0)", + "hexAnnotated": "c2 # tag(2) positive-bignum\n 40 # bytes(0)" + } + }, + { + "name": "tag/bignum-2-text-content-bignum-build", + "input": { + "hex": "c26178" + }, + "config": "standard+bignum", + "build": "bignum", + "expect": { + "hex": "c26178", + "diagnostic": "2(\"x\")", + "annotated": "2(\"x\") / positive-bignum /", + "flat": "2(\"x\")", + "summary": "", + "hexAnnotated": "c2 # tag(2) positive-bignum\n 61 # text(1)\n 78 # \"x\"" + }, + "note": "summarizer error" + }, + { + "name": "tag/bignum-2-text-content-default-build", + "input": { + "hex": "c26178" + }, + "config": "standard", + "build": "default", + "expect": { + "hex": "c26178", + "diagnostic": "2(\"x\")", + "annotated": "2(\"x\")", + "flat": "2(\"x\")", + "summary": "2(\"x\")", + "hexAnnotated": "c2 # tag(2)\n 61 # text(1)\n 78 # \"x\"" + } + } + ] +} diff --git a/tests/vectors/recipes.ts b/tests/vectors/recipes.ts index 77a3024..9e2619d 100644 --- a/tests/vectors/recipes.ts +++ b/tests/vectors/recipes.ts @@ -1,34 +1,32 @@ /** - * Build-agnostic construction recipes for the wire-format vector suites - * (API_REDESIGN_PLAN P1.1). + * Build-agnostic construction recipes for the wire-format vector suites. * - * A `Recipe` is a JSON-serializable description of a construction input - - * everything `cbor()`/`cborData()` accepts today: JS primitives, bigints, - * floats (including NaN/±Infinity/-0), strings, byte arrays, arrays, plain - * objects, JS Map/Set, CborMap, CborSet, CborDate, ByteString, tagged values, - * tag-2/3 bignums, and the two protocol shapes (`toCbor()` / `taggedCbor()`). + * A `Recipe` is a JSON-serializable description of a construction input: JS + * primitives, bigints, floats (including NaN/±Infinity/-0), strings, byte + * arrays, arrays, plain objects, JS Map/Set, CborMap, CborSet, CborDate, + * ByteString, tagged values, tag-2/3 bignums, and the protocol shapes + * (`toCbor()` / `taggedCbor()`). * * Recipes are materialized against a `VectorApi` adapter rather than a - * concrete module, so the SAME recipe can be constructed with two different - * builds of the library: + * concrete module, so the same recipe can be constructed with two builds of + * the library whose public APIs are spelled differently: * - * - the frozen baseline bundle (`tests/baseline/dcbor-baseline.mjs`, - * built from the pre-redesign commit recorded in tests/baseline/README.md) - * - the working tree (`../src`) + * - the baseline bundle (`tests/baseline/dcbor-baseline.mjs`, see + * tests/baseline/README.md), through {@link baselineAdapterFor} + * - the working tree (`../src`), through {@link currentAdapterFor} * - * That is what makes the differential harness (`tests/differential.test.ts`) - * and the committed golden fixtures (`tests/vectors/*.json`) survive the API - * redesign: when Phase 3 renames the public surface, ONLY the working-tree - * adapter in `adapterFor` needs a sibling (write e.g. `redesignedAdapterFor` - * against the new names and point the harnesses' `current` at it). The frozen - * baseline keeps using `adapterFor` - its API never changes - and the recipes, - * corpus, and fixtures stay byte-for-byte identical. - * - * IMPORTANT: recipe semantics are FROZEN. Never change how an existing recipe - * kind materializes; add a new kind instead. The committed fixtures pair - * recipes with expected bytes - changing materialization silently invalidates - * the pairing. + * Recipe semantics are fixed: never change how an existing recipe kind + * materializes; add a new kind instead. The committed fixtures pair recipes + * with expected bytes, so changing materialization silently invalidates the + * pairing. + */ + +/** + * Input shapes the baseline encoded and the working tree rejects with a + * directive `CborError`. Corpus entries and fixtures carrying one are asserted + * to throw rather than to match the baseline's bytes. */ +export type RemovedInputShape = "tag-value-literal" | "tagged-cbor-only"; /** Integer/float/bigint carried as a decimal string so recipes are JSON-safe. */ export type Recipe = @@ -53,27 +51,27 @@ export type Recipe = /** CborMap of `count` entries `i -> "v" + i` (compact count-cliff form). */ | { k: "intmap"; count: number } /** - * Plain JS object. Keys must NOT be array-index-like ("0", "17", …): JS + * Plain JS object. Keys must not be array-index-like ("0", "17", …): JS * enumerates those first in numeric order, silently reordering entries; * the materializer rejects them. Other keys keep insertion order. */ | { k: "obj"; entries: [string, Recipe][] } /** JS Map (insertion order preserved; library sorts canonically). */ | { k: "jsmap"; entries: [Recipe, Recipe][] } - /** JS Set (insertion order preserved; library does NOT sort JS sets). */ + /** JS Set (insertion order preserved; library does not sort JS sets). */ | { k: "jsset"; items: Recipe[] } /** CborMap populated via set() in entry order. */ | { k: "map"; entries: [Recipe, Recipe][] } - /** CborSet.fromArray (canonical sort + dedup). */ + /** CborSet built from the items (canonical sort + dedup). */ | { k: "set"; items: Recipe[] } - /** toTaggedValue(tag, content); tag is a decimal string (may exceed 2^53). */ + /** Tagged value; tag is a decimal string (may exceed 2^53). */ | { k: "tagged"; tag: string; content: Recipe } /** - * Plain object literal shaped exactly `{tag, value}` - the key-sniffing - * input that P3.5 tombstones. Baseline encodes it as a tagged value. + * Plain object literal shaped exactly `{tag, value}`. The baseline encodes + * it as a tagged value; the working tree rejects it (`tag-value-literal`). */ | { k: "tagobjlit"; tag: Recipe; content: Recipe } - /** CborDate.fromTimestamp(Number(seconds)). */ + /** CborDate from epoch seconds, `Number(seconds)`. */ | { k: "date"; seconds: string } /** CborDate.fromString(v). */ | { k: "datestr"; v: string } @@ -86,43 +84,50 @@ export type Recipe = /** Anonymous object implementing only `toCbor()` (ToCbor protocol). */ | { k: "tocbor"; inner: Recipe } /** - * Anonymous object implementing only `taggedCbor()` - the protocol shape - * that P3.7 tombstones (auto-wrap of TaggedCborEncodable). + * Anonymous object implementing only `taggedCbor()`. The baseline + * auto-wrapped it; the working tree rejects it (`tagged-cbor-only`). */ | { k: "taggedproto"; tag: string; inner: Recipe } /** - * Object implementing BOTH `taggedCbor()` and `toCbor()`, each producing - * observably different bytes - freezes the dispatch precedence - * (`taggedCbor` wins today; P3.7 makes `toCbor` the only protocol). + * Object implementing both `taggedCbor()` and `toCbor()`, each producing + * different bytes, so the encoding shows which one dispatch picked + * (`toCbor` in the working tree, `taggedCbor` in the baseline). */ | { k: "bothproto"; tag: string; inner: Recipe } /** * Bare Simple/Float Cbor node `{isCbor, type: 7, value: {type: "Float"}}` * (no methods - exercises the attachMethods passthrough arm). This is the - * ONLY route into the float encoder's own reduction ladder: whole-valued - * plain numbers integer-reduce in cbor() dispatch long before f64CborData - * runs, so the frozen float quirks (fround negative-reduction collisions, - * f32-exact wholes >= 2^32 staying 0xfa floats) are observable only here. + * only route into the float encoder's own reduction ladder: whole-valued + * plain numbers integer-reduce in cbor() dispatch before f64CborData runs, + * so the float quirks (fround negative-reduction collisions, f32-exact + * wholes >= 2^32 staying 0xfa floats) are observable only here. */ | { k: "floatsimple"; v: string } /** Bare methodless Unsigned Cbor node (attachMethods passthrough arm). */ | { k: "rawuint"; v: string } /** - * Bare methodless Negative Cbor node storing the MAGNITUDE-to-encode + * Bare methodless Text Cbor node holding `v` verbatim (no constructor + * normalization). Pins that NFC is applied at encode time, as the reference + * does for `CBORCase::Text`: a decomposed string in the node still encodes + * composed. + */ + | { k: "rawtext"; v: string } + /** + * Bare methodless Negative Cbor node storing the magnitude to encode * (semantic value is -1-v, mirroring the decoder's representation). */ | { k: "rawnegmag"; v: string } /** Malformed bare Cbor node (ByteString type, non-Uint8Array value). */ | { k: "rawbad" } - /** A Symbol input (unsupported by cbor() - frozen Custom throw). */ + /** A Symbol input (unsupported by cbor() - throws Custom). */ | { k: "symbol" } - /** A function input (unsupported by cbor() - frozen Custom throw). */ + /** A function input (unsupported by cbor() - throws Custom). */ | { k: "fn" } /** - * Object with `tag`/`value` INHERITED from its prototype plus own entries. - * The sniffing arm's outer trigger (`"tag" in value`) sees prototype + * Object with `tag`/`value` inherited from its prototype plus own entries. + * The `{tag, value}` check in cbor() (`"tag" in value`) sees prototype * properties but Object.keys does not, so this falls through to the - * plain-object→map branch - frozen boundary behavior for P3.5. + * plain-object→map branch and encodes only the own entries. */ | { k: "protoobj"; protoEntries: [string, Recipe][]; ownEntries: [string, Recipe][] }; @@ -132,11 +137,11 @@ export type Recipe = * whatever the build's public API calls them. */ export interface VectorApi { - /** Full construction + encode: today `cborData(input)`. Throws CborError. */ + /** Full construction + encode. Throws CborError. */ encode(input: unknown): Uint8Array; - /** Strict decode: today `decodeCbor(bytes)`. Throws CborError. */ + /** Strict decode. Throws CborError. */ decode(bytes: Uint8Array): unknown; - /** The polymorphic constructor: today `cbor(input)`. */ + /** The polymorphic constructor, `cbor(input)`. */ makeCbor(input: unknown): unknown; makeMap(entries: [unknown, unknown][]): unknown; makeSet(items: unknown[]): unknown; @@ -151,12 +156,11 @@ export interface VectorApi { } /** - * Adapter for the CURRENT (pre-redesign) public API. Works for both the - * frozen baseline bundle and today's `../src`. When the Phase 3 renames land, - * add a sibling adapter for the new surface and keep this one for the - * baseline - do not edit this function's semantics. + * Adapter for the baseline bundle's public API (`cborData`, `toTaggedValue`, + * `CborSet.fromArray`, `CborDate.fromTimestamp`). Its semantics are fixed with + * the bundle. */ -export function adapterFor(mod: Record): VectorApi { +export function baselineAdapterFor(mod: Record): VectorApi { /* eslint-disable @typescript-eslint/no-explicit-any */ const m = mod as any; return { @@ -187,12 +191,10 @@ export function adapterFor(mod: Record): VectorApi { } /** - * Adapter for the REDESIGNED public API (post-P3 wave). The differential - * harness's `current` side and the golden suite use this; the frozen - * baseline keeps using {@link adapterFor}. Recipes, corpus, and fixtures - * are IDENTICAL for both - only the spellings differ. + * Adapter for the working tree's public API. The differential harness's + * `current` side, the golden suite and the vector generator use it. */ -export function redesignedAdapterFor(mod: Record): VectorApi { +export function currentAdapterFor(mod: Record): VectorApi { /* eslint-disable @typescript-eslint/no-explicit-any */ const m = mod as any; return { @@ -263,7 +265,7 @@ const cycleBytes = (start: number, count: number): Uint8Array => { /** * Construct the input a recipe describes, using `api`'s build of the library. - * May throw that build's CborError (e.g. `date` with a non-finite timestamp, + * May throw that build's CborError (e.g. `date` with an infinite timestamp, * `biguint` with a negative value) - callers wanting an outcome use * {@link encodeOutcome}, which captures those uniformly. */ @@ -319,7 +321,7 @@ export function materialize(recipe: Recipe, api: VectorApi): unknown { case "tagged": return api.makeTagged(tagFromString(recipe.tag), materialize(recipe.content, api)); case "tagobjlit": - // Exactly the two own keys {tag, value} - the P3.5 sniffing shape. + // Exactly the two own keys {tag, value}. return { tag: materialize(recipe.tag, api), value: materialize(recipe.content, api) }; case "date": return api.makeDate(Number(recipe.seconds)); @@ -356,6 +358,8 @@ export function materialize(recipe: Recipe, api: VectorApi): unknown { const big = BigInt(recipe.v); return { isCbor: true, type: 0, value: big <= MAX_SAFE ? Number(big) : big }; } + case "rawtext": + return { isCbor: true, type: 3, value: recipe.v }; case "rawnegmag": { const big = BigInt(recipe.v); return { isCbor: true, type: 1, value: big <= MAX_SAFE ? Number(big) : big }; @@ -403,6 +407,21 @@ export function encodeOutcome(api: VectorApi, recipe: Recipe): EncodeOutcome { } } +/** + * The message of the CborError a decode raises, or `undefined` when the + * bytes decode. Used by the golden fixtures: the reference's error `Display` + * text is part of the contract (`tests/rust-validation` compares it). + */ +export function decodeErrorMessage(api: VectorApi, bytes: Uint8Array): string | undefined { + try { + api.decode(bytes); + return undefined; + } catch (e) { + if (api.errorCode(e) === undefined) throw e; + return (e as Error).message; + } +} + /** Decode + re-encode, capturing the build's CborError as a staged code. */ export function decodeOutcome(api: VectorApi, bytes: Uint8Array): DecodeOutcome { let decoded: unknown; diff --git a/tests/vectors/uint-corpus.ts b/tests/vectors/uint-corpus.ts new file mode 100644 index 0000000..274ca6a --- /dev/null +++ b/tests/vectors/uint-corpus.ts @@ -0,0 +1,50 @@ +/** + * Curated UNSIGNED-EXTRACTION corpus. + * + * Each row decodes `hex` and extracts it into a fixed-width unsigned + * integer: `expectUnsigned(cbor, { width, wrapNegative: true })` here, + * `u8`/`u16`/`u32`/`u64::try_from(CBOR)` in the Rust harness. The + * reference wraps a negative integer whose magnitude fits the width + * (RUST_DIVERGENCES.md §1.1) and rejects everything else with + * `OutOfRange`; non-integers are `WrongType`. + */ + +export interface UintCorpusEntry { + name: string; + hex: string; + width: 8 | 16 | 32 | 64; + note?: string; +} + +const row = (name: string, hex: string, width: 8 | 16 | 32 | 64, note?: string): UintCorpusEntry => + note === undefined ? { name, hex, width } : { name, hex, width, note }; + +export const uintCorpus: UintCorpusEntry[] = [ + row("u8/-1-wraps-255", "20", 8), + row("u8/-256-wraps-0", "38ff", 8), + row("u8/-257-out-of-range", "390100", 8), + row("u8/256-out-of-range", "190100", 8), + row("u8/255", "18ff", 8), + row("u8/0", "00", 8), + row("u16/-1-wraps-65535", "20", 16), + row("u16/65536-out-of-range", "1a00010000", 16), + row("u32/-1-wraps-4294967295", "20", 32), + row("u32/-2^32-wraps-0", "3affffffff", 32), + row("u32/-2^32-1-out-of-range", "3b0000000100000000", 32), + row("u32/2^32-out-of-range", "1b0000000100000000", 32), + row("u64/-2^64-wraps-0", "3bffffffffffffffff", 64, "the 65-bit negative"), + row("u64/-1-wraps-max", "20", 64), + row("u64/max", "1bffffffffffffffff", 64), + row("u64/2^53", "1b0020000000000000", 64), + row("u8/float-wrong-type", "f93e00", 8), + row("u8/text-wrong-type", "6161", 8), + row("u64/tagged-wrong-type", "c100", 64), +]; + +{ + const seen = new Set(); + for (const { name } of uintCorpus) { + if (seen.has(name)) throw new Error(`duplicate uint-corpus name: ${name}`); + seen.add(name); + } +} diff --git a/tests/vectors/uint-vectors.json b/tests/vectors/uint-vectors.json new file mode 100644 index 0000000..105a877 --- /dev/null +++ b/tests/vectors/uint-vectors.json @@ -0,0 +1,179 @@ +{ + "//": "GENERATED by scripts/generate-vectors.ts - do not edit by hand. Review regenerated expectations before committing them.", + "sourceCommit": "02766ae64e450096afd888380a8f80bcf24c2511", + "count": 19, + "vectors": [ + { + "name": "u8/-1-wraps-255", + "hex": "20", + "width": 8, + "expect": { + "ok": true, + "value": "255" + } + }, + { + "name": "u8/-256-wraps-0", + "hex": "38ff", + "width": 8, + "expect": { + "ok": true, + "value": "0" + } + }, + { + "name": "u8/-257-out-of-range", + "hex": "390100", + "width": 8, + "expect": { + "ok": false, + "code": "OutOfRange" + } + }, + { + "name": "u8/256-out-of-range", + "hex": "190100", + "width": 8, + "expect": { + "ok": false, + "code": "OutOfRange" + } + }, + { + "name": "u8/255", + "hex": "18ff", + "width": 8, + "expect": { + "ok": true, + "value": "255" + } + }, + { + "name": "u8/0", + "hex": "00", + "width": 8, + "expect": { + "ok": true, + "value": "0" + } + }, + { + "name": "u16/-1-wraps-65535", + "hex": "20", + "width": 16, + "expect": { + "ok": true, + "value": "65535" + } + }, + { + "name": "u16/65536-out-of-range", + "hex": "1a00010000", + "width": 16, + "expect": { + "ok": false, + "code": "OutOfRange" + } + }, + { + "name": "u32/-1-wraps-4294967295", + "hex": "20", + "width": 32, + "expect": { + "ok": true, + "value": "4294967295" + } + }, + { + "name": "u32/-2^32-wraps-0", + "hex": "3affffffff", + "width": 32, + "expect": { + "ok": true, + "value": "0" + } + }, + { + "name": "u32/-2^32-1-out-of-range", + "hex": "3b0000000100000000", + "width": 32, + "expect": { + "ok": false, + "code": "OutOfRange" + } + }, + { + "name": "u32/2^32-out-of-range", + "hex": "1b0000000100000000", + "width": 32, + "expect": { + "ok": false, + "code": "OutOfRange" + } + }, + { + "name": "u64/-2^64-wraps-0", + "hex": "3bffffffffffffffff", + "width": 64, + "expect": { + "ok": true, + "value": "0" + }, + "note": "the 65-bit negative" + }, + { + "name": "u64/-1-wraps-max", + "hex": "20", + "width": 64, + "expect": { + "ok": true, + "value": "18446744073709551615" + } + }, + { + "name": "u64/max", + "hex": "1bffffffffffffffff", + "width": 64, + "expect": { + "ok": true, + "value": "18446744073709551615" + } + }, + { + "name": "u64/2^53", + "hex": "1b0020000000000000", + "width": 64, + "expect": { + "ok": true, + "value": "9007199254740992" + } + }, + { + "name": "u8/float-wrong-type", + "hex": "f93e00", + "width": 8, + "expect": { + "ok": false, + "code": "WrongType" + } + }, + { + "name": "u8/text-wrong-type", + "hex": "6161", + "width": 8, + "expect": { + "ok": false, + "code": "WrongType" + } + }, + { + "name": "u64/tagged-wrong-type", + "hex": "c100", + "width": 64, + "expect": { + "ok": false, + "code": "WrongType" + } + } + ] +} diff --git a/tests/walk.test.ts b/tests/walk.test.ts index 5e173cc..5bd2088 100644 --- a/tests/walk.test.ts +++ b/tests/walk.test.ts @@ -1,40 +1,6 @@ /** - * Walk Module Integration Tests - 1:1 translation from Rust's tests/walk.rs - * - * This file contains comprehensive integration tests for the `walk` module. - * - * ## Test Coverage - * - * ### Basic Functionality - * - **test_traversal_counts**: Verifies correct visit counts for different - * CBOR structures (arrays, maps, tagged values, nested structures) - * - **test_visitor_state_threading**: Tests that visitor state is properly - * maintained through traversal - * - **test_primitive_values**: Ensures primitive values are handled correctly - * - **test_empty_structures**: Tests behavior with empty arrays and maps - * - * ### Traversal Semantics - * - **test_traversal_order_and_edge_types**: Validates the order of visits and - * correct edge type labeling - * - **test_map_keyvalue_semantics**: Verifies that map key-value pairs are - * visited both as semantic units and individually - * - **test_tagged_value_traversal**: Tests traversal of tagged values and - * nested tagged structures - * - * ### Advanced Features - * - **test_depth_limited_traversal**: Tests depth-limited traversal using the - * level parameter - * - **test_early_termination**: Demonstrates controlled termination using the - * stop flag to prevent descent into children - * - **test_stop_flag_prevents_descent**: Verifies that the stop flag - * consistently prevents descent into children while allowing sibling - * traversal - * - * ### Real-World Usage - * - **test_text_extraction**: Extracts all text strings from a complex nested - * structure - * - **test_real_world_document**: Tests traversal of a realistic JSON-like - * document structure converted to CBOR + * Walk tests, ported from Rust's tests/walk.rs and the unit tests in + * src/walk.rs. */ import type { CborInput } from "../src"; @@ -66,7 +32,6 @@ function countVisits(cborValue: CborInput): number { } describe("walk tests", () => { - /// Test basic traversal counts for different CBOR structures test("test_traversal_counts", () => { // Simple array const array = [1, 2, 3]; @@ -108,7 +73,6 @@ describe("walk tests", () => { expect(count4).toBe(12); }); - /// Test that visitor state is properly threaded through traversal test("test_visitor_state_threading", () => { const array = [1, 2, 3, 4, 5]; @@ -137,7 +101,6 @@ describe("walk tests", () => { expect(evenCount).toBe(2); // 2 and 4 are even }); - /// Test early termination using visitor pattern test("test_early_termination", () => { // Test shows that stop flag prevents descent into children but doesn't // abort entire walk @@ -219,7 +182,6 @@ describe("walk tests", () => { expect(level2AfterThird.length).toBe(0); }); - /// Test depth-limited traversal using level parameter test("test_depth_limited_traversal", () => { // Create deeply nested structure const level3 = new CborMap(); @@ -254,7 +216,6 @@ describe("walk tests", () => { expect(elementsByLevel[3] || 0).toBe(0); // No visits at level 3 due to stop }); - /// Test text extraction from complex CBOR structures test("test_text_extraction", () => { // Create a complex structure with text at various levels const metadata = new CborMap(); @@ -318,7 +279,6 @@ describe("walk tests", () => { expect(texts).toContain("tags"); }); - /// Test traversal order and edge types test("test_traversal_order_and_edge_types", () => { const map = new CborMap(); map.set("a", [1, 2]); @@ -366,7 +326,6 @@ describe("walk tests", () => { expect(hasArrayElement1).toBe(true); }); - /// Test tagged value traversal test("test_tagged_value_traversal", () => { // Create nested tagged values const innerTagged = taggedValue(123, [1, 2, 3]); @@ -417,7 +376,6 @@ describe("walk tests", () => { if (edge5.type === "array_element") expect(edge5.index).toBe(2); }); - /// Test map key-value semantics test("test_map_keyvalue_semantics", () => { const map = new CborMap(); map.set("simple", 42); @@ -450,7 +408,6 @@ describe("walk tests", () => { expect(individualCount).toBe(4); }); - /// Test stop flag prevents descent consistently test("test_stop_flag_prevents_descent", () => { const nested = [ [1, 2, 3], // Index 0: prevent descent into this @@ -509,7 +466,6 @@ describe("walk tests", () => { expect(hasLevel2From789).toBe(true); }); - /// Test empty structures test("test_empty_structures", () => { // Empty array const emptyArray: number[] = []; @@ -522,7 +478,6 @@ describe("walk tests", () => { expect(count2).toBe(1); // Just the root }); - /// Test primitive values test("test_primitive_values", () => { const primitives = [42, "hello", 3.2222, true, null]; @@ -532,7 +487,6 @@ describe("walk tests", () => { } }); - /// Test real-world document structure test("test_real_world_document", () => { // Simulate a JSON-like document converted to CBOR const person = new CborMap(); @@ -610,8 +564,7 @@ describe("walk tests", () => { expect(strings).toContain("languages"); }); - /// Root array: visit count and exact positional edge ordering - /// (port of src/walk.rs `test_walk_array`) + // Port of src/walk.rs `test_walk_array`: visit count and positional edges. test("test_walk_array", () => { const edges: EdgeTypeVariant[] = []; let count = 0; @@ -638,8 +591,7 @@ describe("walk tests", () => { } }); - /// Single-level tagged value: visit count and edge sequence - /// (port of src/walk.rs `test_walk_tagged`) + // Port of src/walk.rs `test_walk_tagged`. test("test_walk_tagged", () => { const tagged = taggedValue(0, "2023-01-01T00:00:00Z"); @@ -664,8 +616,7 @@ describe("walk tests", () => { expect(edges[1]?.type).toBe("tagged_content"); // Content }); - /// Nested map-with-array visit count (port of src/walk.rs - /// `test_walk_nested_structure`) + // Port of src/walk.rs `test_walk_nested_structure`. test("test_walk_nested_structure", () => { const map = new CborMap(); map.set("numbers", [1, 2, 3]); @@ -676,7 +627,7 @@ describe("walk tests", () => { expect(countVisits(map)).toBe(10); }); - /// Edge-type labels (port of src/walk.rs `test_edge_type_labels`) + // Port of src/walk.rs `test_edge_type_labels`. test("test_edge_type_labels", () => { expect(edgeLabel({ type: "none" })).toBeUndefined(); expect(edgeLabel({ type: "array_element", index: 5 })).toBe("arr[5]"); diff --git a/vitest.config.ts b/vitest.config.ts index 44cd750..4204ce6 100644 --- a/vitest.config.ts +++ b/vitest.config.ts @@ -12,7 +12,7 @@ export default defineConfig({ reportsDirectory: "coverage", include: ["src/**/*.ts"], exclude: ["src/**/*.d.ts", "src/index.ts"], - // Raise-only floors. Seed from the first measured run; never lower. + // Raise-only floors: raise them as coverage grows, never lower them. thresholds: { statements: 70, branches: 66,