From adc56c5e5274c42d68569c315391de681af0b41e Mon Sep 17 00:00:00 2001 From: Sankalp Gilda Date: Sat, 19 Sep 2026 17:16:26 -0400 Subject: [PATCH 1/2] refactor(canonical): delegate JCS to a from-spec canonicalizer with vectors Replace the hand-written RFC 8785 walker in atomic-canonical/src/jcs.rs with a call into serde_json_canonicalizer. Member ordering and string escaping were already correct; number formatting was not. Section 3.2.2.3 requires the ECMAScript Number::toString algorithm, and serde_json's formatter crosses between decimal and exponent notation at different magnitudes and writes negative zero as -0.0, so two conforming implementations hashed the same logical document to different digests. The delegate formats through ryu_js, which is the variant the section names, and serializes through an explicit heap stack rather than the call stack. canonicalize keeps its signature, so no call site changes anywhere in the workspace. Eleven measured cases are pinned as fixtures under atomic-canonical/tests/vectors/ with a harness in tests/jcs_vectors.rs. Three conformance cases this entry point cannot decide are recorded in atomic-canonical/tests/vectors/INGEST-BOUNDARY.md rather than tested here: a repeated object member is gone before canonicalize is reached, and RFC 8785 admits both an integer past 2^53 and a fractional number. All of them want a strict decoder on the raw bytes at the boundary where documents arrive. --- Cargo.lock | 18 +++ atomic-canonical/Cargo.toml | 1 + atomic-canonical/src/jcs.rs | 81 +++++-------- atomic-canonical/tests/jcs_vectors.rs | 113 ++++++++++++++++++ .../tests/vectors/INGEST-BOUNDARY.md | 30 +++++ .../depth-at-the-cap-is-canonicalized.json | 8 ++ .../tests/vectors/number-decimal-2pow68.json | 8 ++ .../number-decimal-999999999999999700000.json | 8 ++ .../number-decimal-999999999999999900000.json | 8 ++ ...mber-decimal-below-exponent-threshold.json | 8 ++ ...number-exponent-1.0000000000000001e23.json | 8 ++ .../number-exponent-9.999999999999997e-7.json | 8 ++ .../number-exponent-9.999999999999997e22.json | 8 ++ .../number-negative-small-decimal.json | 8 ++ .../tests/vectors/number-negative-zero.json | 8 ++ .../vectors/number-rounded-to-its-double.json | 8 ++ 16 files changed, 280 insertions(+), 51 deletions(-) create mode 100644 atomic-canonical/tests/jcs_vectors.rs create mode 100644 atomic-canonical/tests/vectors/INGEST-BOUNDARY.md create mode 100644 atomic-canonical/tests/vectors/depth-at-the-cap-is-canonicalized.json create mode 100644 atomic-canonical/tests/vectors/number-decimal-2pow68.json create mode 100644 atomic-canonical/tests/vectors/number-decimal-999999999999999700000.json create mode 100644 atomic-canonical/tests/vectors/number-decimal-999999999999999900000.json create mode 100644 atomic-canonical/tests/vectors/number-decimal-below-exponent-threshold.json create mode 100644 atomic-canonical/tests/vectors/number-exponent-1.0000000000000001e23.json create mode 100644 atomic-canonical/tests/vectors/number-exponent-9.999999999999997e-7.json create mode 100644 atomic-canonical/tests/vectors/number-exponent-9.999999999999997e22.json create mode 100644 atomic-canonical/tests/vectors/number-negative-small-decimal.json create mode 100644 atomic-canonical/tests/vectors/number-negative-zero.json create mode 100644 atomic-canonical/tests/vectors/number-rounded-to-its-double.json diff --git a/Cargo.lock b/Cargo.lock index a4bd3044..6a5dd471 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -169,6 +169,7 @@ dependencies = [ "data-encoding", "serde", "serde_json", + "serde_json_canonicalizer", "thiserror 1.0.69", ] @@ -2508,6 +2509,12 @@ version = "1.0.23" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "9774ba4a74de5f7b1c1451ed6cd5285a32eddb5cccb8cc655a4e50009e06477f" +[[package]] +name = "ryu-js" +version = "1.0.3" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "04d056b875a9d2e6cb9a61d127afee9ac5999b9f87bcb32079d1318e505be714" + [[package]] name = "same-file" version = "1.0.6" @@ -2629,6 +2636,17 @@ dependencies = [ "zmij", ] +[[package]] +name = "serde_json_canonicalizer" +version = "0.3.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "fe52319a927259afbfa5180c5157cd8167edfd3e8c254f9558c7fef44c5649f2" +dependencies = [ + "ryu-js", + "serde", + "serde_json", +] + [[package]] name = "serde_spanned" version = "0.6.9" diff --git a/atomic-canonical/Cargo.toml b/atomic-canonical/Cargo.toml index 58ceac78..c6c48ce5 100644 --- a/atomic-canonical/Cargo.toml +++ b/atomic-canonical/Cargo.toml @@ -16,4 +16,5 @@ bs58 = { workspace = true } chrono = { workspace = true } thiserror = { workspace = true } data-encoding = { workspace = true } +serde_json_canonicalizer = "0.3" atomic-identity = { workspace = true } diff --git a/atomic-canonical/src/jcs.rs b/atomic-canonical/src/jcs.rs index b1ec4d09..d0682a2b 100644 --- a/atomic-canonical/src/jcs.rs +++ b/atomic-canonical/src/jcs.rs @@ -9,63 +9,42 @@ //! hash (`hash.rs`) and the Data Integrity proof (`proof.rs`) go through //! `canonicalize`, so the two can never drift. //! -//! Object keys are sorted by UTF-16 code units as RFC 8785 §3.2.3 specifies -//! (not by UTF-8 bytes — the two differ once keys leave the BMP, e.g. an -//! emoji key sorts after `\u{ff61}` in UTF-8 but before it in UTF-16). +//! # The algorithm is delegated rather than written here //! -//! Scope note: numbers are emitted via `serde_json`'s formatter, which -//! matches RFC 8785 for the integer values our vocabulary admits; the full -//! ECMAScript number-to-string algorithm is only needed if floating-point -//! payloads are ever admitted. +//! The bytes come from `serde_json_canonicalizer`. Two of the three parts of +//! RFC 8785 are straightforward to write by hand and the third is not: +//! +//! * Object keys sort by UTF-16 code units as §3.2.3 specifies, not by UTF-8 +//! bytes — the two differ once a key leaves the BMP, e.g. an emoji key sorts +//! after `\u{ff61}` in UTF-8 but before it in UTF-16. +//! * Strings take the short form for the seven named escapes and lowercase +//! `\u00xx` for the rest of C0 (§3.2.2.2). +//! * **Numbers take the ECMAScript `Number::toString` algorithm** (§3.2.2.3), +//! which `serde_json`'s formatter is not. `serde_json` writes `-0.0` where the +//! algorithm writes `0`, keeps a `.0` on an integer-valued double where the +//! algorithm drops it, and crosses between decimal and exponent notation at +//! different magnitudes: `1e-6` for `0.000001`, `2.9514790517935283e+20` for +//! `295147905179352830000`. The delegate formats through `ryu_js`, which is +//! the ECMAScript variant the section names, and it serializes through an +//! explicit heap stack rather than the call stack. The cases that used to +//! diverge are pinned in `tests/vectors/`. +//! +//! Scope note: the entry point takes an already-parsed [`Value`], so faults that +//! only exist in the wire bytes cannot be decided here — a repeated object +//! member is gone before this function is called, and RFC 8785 admits an integer +//! past 2^53 by rounding it to its double. Those belong to a strict decoder at +//! the boundary where the bytes arrive; `tests/vectors/INGEST-BOUNDARY.md` names +//! the cases and what closes them. use serde_json::Value; /// Canonicalize a JSON value into its RFC-8785 string form. pub fn canonicalize(value: &Value) -> String { - let mut out = String::new(); - write_value(&mut out, value); - out -} - -fn write_value(out: &mut String, value: &Value) { - match value { - Value::Null => out.push_str("null"), - Value::Bool(true) => out.push_str("true"), - Value::Bool(false) => out.push_str("false"), - Value::Number(n) => out.push_str(&n.to_string()), - Value::String(s) => write_json_string(out, s), - Value::Array(items) => { - out.push('['); - for (i, item) in items.iter().enumerate() { - if i > 0 { - out.push(','); - } - write_value(out, item); - } - out.push(']'); - } - Value::Object(map) => { - let mut keys: Vec<&String> = map.keys().collect(); - keys.sort_by(|a, b| a.encode_utf16().cmp(b.encode_utf16())); - out.push('{'); - for (i, key) in keys.iter().enumerate() { - if i > 0 { - out.push(','); - } - write_json_string(out, key); - out.push(':'); - write_value(out, &map[*key]); - } - out.push('}'); - } - } -} - -/// Emit a JSON string with standard escaping. `serde_json` produces a valid, -/// minimally-escaped JSON string literal (quotes included), which matches JCS -/// for the ASCII content in our records. -fn write_json_string(out: &mut String, s: &str) { - out.push_str(&serde_json::to_string(s).expect("string serialization is infallible")); + // Infallible for a `Value`: there is no writer to fail against, every + // member name is already a Rust `String`, and the delegate's only other + // error path is a non-finite float, which `Value` cannot hold. This mirrors + // the expectation the hand-written string helper carried before it. + serde_json_canonicalizer::to_string(value).expect("canonicalizing a Value is infallible") } #[cfg(test)] diff --git a/atomic-canonical/tests/jcs_vectors.rs b/atomic-canonical/tests/jcs_vectors.rs new file mode 100644 index 00000000..02ccc69e --- /dev/null +++ b/atomic-canonical/tests/jcs_vectors.rs @@ -0,0 +1,113 @@ +//! Canonicalization vectors, one fixture per divergence. +//! +//! Each file in `tests/vectors/` carries an input document, the divergence it +//! pins, and the canonical form RFC 8785 requires. The number cases are Appendix +//! B rows: the canonical text of a double is one string and no other, so a +//! canonicalizer that writes `1e-6` where the algorithm writes `0.000001` +//! produces a different hash for the same value, and a second implementation +//! then rejects a proof this one accepts. +//! +//! `tests/vectors/INGEST-BOUNDARY.md` lists the conformance cases this entry +//! point cannot decide, because it receives an already-parsed value rather than +//! the bytes. + +use std::fs; +use std::path::{Path, PathBuf}; + +use atomic_canonical::jcs; +use serde_json::Value; + +fn vector_dir() -> PathBuf { + Path::new(env!("CARGO_MANIFEST_DIR")).join("tests/vectors") +} + +fn fixture(name: &str) -> Value { + let path = vector_dir().join(format!("{name}.json")); + let raw = fs::read_to_string(&path).unwrap_or_else(|e| panic!("read {path:?}: {e}")); + serde_json::from_str(&raw).unwrap_or_else(|e| panic!("parse {path:?}: {e}")) +} + +/// Parse the fixture's input the way any caller receiving JSON does, canonicalize +/// it, and require the exact bytes. +fn assert_canonical(name: &str) { + let f = fixture(name); + let text = f["input"].as_str().expect("fixture carries an input string"); + let value: Value = serde_json::from_str(text).expect("fixture input is valid JSON"); + let expected = f["expect"]["canonical"] + .as_str() + .expect("fixture expects a canonical form"); + assert_eq!( + jcs::canonicalize(&value), + expected, + "{name} ({}): {}", + f["vector"].as_str().unwrap_or("?"), + f["divergence"].as_str().unwrap_or("") + ); +} + +#[test] +fn number_decimal_below_exponent_threshold() { + assert_canonical("number-decimal-below-exponent-threshold"); +} + +#[test] +fn number_decimal_2pow68() { + assert_canonical("number-decimal-2pow68"); +} + +#[test] +fn number_decimal_999999999999999700000() { + assert_canonical("number-decimal-999999999999999700000"); +} + +#[test] +fn number_decimal_999999999999999900000() { + assert_canonical("number-decimal-999999999999999900000"); +} + +#[test] +fn number_negative_small_decimal() { + assert_canonical("number-negative-small-decimal"); +} + +#[test] +fn number_negative_zero() { + assert_canonical("number-negative-zero"); +} + +#[test] +fn number_exponent_9_999999999999997e22() { + assert_canonical("number-exponent-9.999999999999997e22"); +} + +#[test] +fn number_exponent_1_0000000000000001e23() { + assert_canonical("number-exponent-1.0000000000000001e23"); +} + +#[test] +fn number_exponent_9_999999999999997e_minus_7() { + assert_canonical("number-exponent-9.999999999999997e-7"); +} + +#[test] +fn number_rounded_to_its_double() { + assert_canonical("number-rounded-to-its-double"); +} + +/// The accept half of the depth condition: a document at 128 containers +/// canonicalizes, and the delegate walks it on the heap rather than the call +/// stack. The refusal half needs a fallible boundary -- see +/// `tests/vectors/INGEST-BOUNDARY.md`. +#[test] +fn depth_at_the_cap_is_canonicalized() { + let f = fixture("depth-at-the-cap-is-canonicalized"); + assert!(f["input_generated"].as_str().is_some()); + + let mut value = Value::Null; + for _ in 0..128 { + value = Value::Array(vec![value]); + } + let canonical = jcs::canonicalize(&value); + assert_eq!(canonical, format!("{}null{}", "[".repeat(128), "]".repeat(128))); +} diff --git a/atomic-canonical/tests/vectors/INGEST-BOUNDARY.md b/atomic-canonical/tests/vectors/INGEST-BOUNDARY.md new file mode 100644 index 00000000..20f86019 --- /dev/null +++ b/atomic-canonical/tests/vectors/INGEST-BOUNDARY.md @@ -0,0 +1,30 @@ +# Vectors this entry point cannot decide + +`jcs::canonicalize` takes an already-parsed `serde_json::Value`. Three conformance +vectors ask for a refusal that no function of that shape can give, because the +fault either vanished during the parse or is not a fault under RFC 8785 at all. +They are listed here rather than dropped, so the follow-up has its brief in-tree. + +| vector | condition | what it asks for | why not here | +| --- | --- | --- | --- | +| `v0f4f2093061d303f` | duplicate member | reject `{"a":1,"a":2}` | `serde_json` keeps one of the two members while parsing, so the repeat is gone before `canonicalize` is called. Measured: the document canonicalizes to `{"a":2}`, and `{"a":2,"a":1}` to `{"a":1}` -- two wire documents, two canonical forms, and a signature over either verifies. | +| `v679f56481420e45a` | unsafe integer | reject `9007199254740993` | RFC 8785 defers number formatting to ECMAScript, which has one numeric type, so the specification *admits* the token and writes the double it rounds to. Refusing it is the RFC 7493 I-JSON profile, which is a stricter profile rather than RFC 8785 itself. Pinned as an accept in `number-rounded-to-its-double.json`. | +| `v97f5d8777e514257` | non-integer in a signed field | reject `0.7` | Same shape: RFC 8785 admits a fractional number. Refusing one is a field-level profile decision, not a canonicalization rule. | + +A fourth case is half-covered. `vd94ac70c9f0d84bf` asks a canonicalizer to refuse +a document nested one container past a stated cap with a catchable error. The +accept half is pinned in `depth-at-the-cap-is-canonicalized.json`; the refusal +half needs a fallible boundary, and an infallible `canonicalize(&Value) -> String` +has nowhere to put it. + +## What closes all four + +A strict decoder on the raw bytes, ahead of this function: it refuses a repeated +member, caps nesting at 128 with an error rather than a stack walk, refuses a +string that is not a sequence of Unicode scalar values, and optionally applies the +I-JSON safe-integer profile. `jcs-admit` on crates.io does exactly that and then +hands canonical output to the same `serde_json_canonicalizer` this file already +uses, so adopting it adds refusals without changing a single byte of output. + +The place it belongs is wherever a document arrives as bytes rather than as a +value built in-process. diff --git a/atomic-canonical/tests/vectors/depth-at-the-cap-is-canonicalized.json b/atomic-canonical/tests/vectors/depth-at-the-cap-is-canonicalized.json new file mode 100644 index 00000000..c0a3a890 --- /dev/null +++ b/atomic-canonical/tests/vectors/depth-at-the-cap-is-canonicalized.json @@ -0,0 +1,8 @@ +{ + "vector": "vd94ac70c9f0d84bf (accept half)", + "divergence": "a document nested 128 containers deep canonicalizes, which is the accept side of the depth condition", + "input_generated": "128 nested arrays around null", + "expect": { + "canonical_shape": "128 open brackets, null, 128 close brackets" + } +} diff --git a/atomic-canonical/tests/vectors/number-decimal-2pow68.json b/atomic-canonical/tests/vectors/number-decimal-2pow68.json new file mode 100644 index 00000000..57d98b14 --- /dev/null +++ b/atomic-canonical/tests/vectors/number-decimal-2pow68.json @@ -0,0 +1,8 @@ +{ + "vector": "v362600d69975d14c", + "divergence": "decimal-to-exponent threshold: a magnitude below 1e21 stays decimal", + "input": "{\"n\":295147905179352830000}", + "expect": { + "canonical": "{\"n\":295147905179352830000}" + } +} diff --git a/atomic-canonical/tests/vectors/number-decimal-999999999999999700000.json b/atomic-canonical/tests/vectors/number-decimal-999999999999999700000.json new file mode 100644 index 00000000..11a1c1b0 --- /dev/null +++ b/atomic-canonical/tests/vectors/number-decimal-999999999999999700000.json @@ -0,0 +1,8 @@ +{ + "vector": "vaa33c34f7bd1058b", + "divergence": "decimal notation retained just below the 1e21 threshold", + "input": "{\"n\":999999999999999700000}", + "expect": { + "canonical": "{\"n\":999999999999999700000}" + } +} diff --git a/atomic-canonical/tests/vectors/number-decimal-999999999999999900000.json b/atomic-canonical/tests/vectors/number-decimal-999999999999999900000.json new file mode 100644 index 00000000..6c188bf1 --- /dev/null +++ b/atomic-canonical/tests/vectors/number-decimal-999999999999999900000.json @@ -0,0 +1,8 @@ +{ + "vector": "vb177ef3b3a945a72", + "divergence": "decimal notation retained for the last double below 1e21", + "input": "{\"n\":999999999999999900000}", + "expect": { + "canonical": "{\"n\":999999999999999900000}" + } +} diff --git a/atomic-canonical/tests/vectors/number-decimal-below-exponent-threshold.json b/atomic-canonical/tests/vectors/number-decimal-below-exponent-threshold.json new file mode 100644 index 00000000..96129517 --- /dev/null +++ b/atomic-canonical/tests/vectors/number-decimal-below-exponent-threshold.json @@ -0,0 +1,8 @@ +{ + "vector": "vaf1b67f038ffff84", + "divergence": "decimal-to-exponent threshold: 1e-6 is written in decimal notation", + "input": "{\"n\":0.000001}", + "expect": { + "canonical": "{\"n\":0.000001}" + } +} diff --git a/atomic-canonical/tests/vectors/number-exponent-1.0000000000000001e23.json b/atomic-canonical/tests/vectors/number-exponent-1.0000000000000001e23.json new file mode 100644 index 00000000..21bb9d5e --- /dev/null +++ b/atomic-canonical/tests/vectors/number-exponent-1.0000000000000001e23.json @@ -0,0 +1,8 @@ +{ + "vector": "vb7dd5fb8dcb6e345", + "divergence": "seventeen significant digits kept where sixteen would not round-trip", + "input": "{\"n\":1.0000000000000001e+23}", + "expect": { + "canonical": "{\"n\":1.0000000000000001e+23}" + } +} diff --git a/atomic-canonical/tests/vectors/number-exponent-9.999999999999997e-7.json b/atomic-canonical/tests/vectors/number-exponent-9.999999999999997e-7.json new file mode 100644 index 00000000..4f2e4350 --- /dev/null +++ b/atomic-canonical/tests/vectors/number-exponent-9.999999999999997e-7.json @@ -0,0 +1,8 @@ +{ + "vector": "vf2f6991c55096b1a", + "divergence": "shortest round-tripping digits below the 1e-6 threshold", + "input": "{\"n\":9.999999999999997e-7}", + "expect": { + "canonical": "{\"n\":9.999999999999997e-7}" + } +} diff --git a/atomic-canonical/tests/vectors/number-exponent-9.999999999999997e22.json b/atomic-canonical/tests/vectors/number-exponent-9.999999999999997e22.json new file mode 100644 index 00000000..50f2ebfa --- /dev/null +++ b/atomic-canonical/tests/vectors/number-exponent-9.999999999999997e22.json @@ -0,0 +1,8 @@ +{ + "vector": "v843566cf72cd33c3", + "divergence": "shortest round-tripping digits in exponent notation", + "input": "{\"n\":9.999999999999997e+22}", + "expect": { + "canonical": "{\"n\":9.999999999999997e+22}" + } +} diff --git a/atomic-canonical/tests/vectors/number-negative-small-decimal.json b/atomic-canonical/tests/vectors/number-negative-small-decimal.json new file mode 100644 index 00000000..afc61d9d --- /dev/null +++ b/atomic-canonical/tests/vectors/number-negative-small-decimal.json @@ -0,0 +1,8 @@ +{ + "vector": "vd2659c36ece7eb47", + "divergence": "a negative magnitude above 1e-6 stays decimal", + "input": "{\"n\":-0.0000033333333333333333}", + "expect": { + "canonical": "{\"n\":-0.0000033333333333333333}" + } +} diff --git a/atomic-canonical/tests/vectors/number-negative-zero.json b/atomic-canonical/tests/vectors/number-negative-zero.json new file mode 100644 index 00000000..ad01f931 --- /dev/null +++ b/atomic-canonical/tests/vectors/number-negative-zero.json @@ -0,0 +1,8 @@ +{ + "vector": "v9a8b364b8bc121de", + "divergence": "negative zero is written as 0, and -0.0 is not valid ECMAScript output at all", + "input": "{\"n\":-0.0}", + "expect": { + "canonical": "{\"n\":0}" + } +} diff --git a/atomic-canonical/tests/vectors/number-rounded-to-its-double.json b/atomic-canonical/tests/vectors/number-rounded-to-its-double.json new file mode 100644 index 00000000..c23749de --- /dev/null +++ b/atomic-canonical/tests/vectors/number-rounded-to-its-double.json @@ -0,0 +1,8 @@ +{ + "vector": "v679f56481420e45a", + "divergence": "RFC 8785 treats every number as a double, so 2^53+1 canonicalizes to the double it rounds to", + "input": "{\"n\":9007199254740993}", + "expect": { + "canonical": "{\"n\":9007199254740992}" + } +} From af313df622e3b130169e48979456a3d91d4b87ff Mon Sep 17 00:00:00 2001 From: Sankalp Gilda Date: Fri, 25 Sep 2026 12:27:32 -0400 Subject: [PATCH 2/2] style(canonical): apply rustfmt to the JCS vector tests The Format job runs cargo fmt --all -- --check, which rejected two statements in atomic-canonical/tests/jcs_vectors.rs. No behaviour change. --- atomic-canonical/tests/jcs_vectors.rs | 9 +++++++-- 1 file changed, 7 insertions(+), 2 deletions(-) diff --git a/atomic-canonical/tests/jcs_vectors.rs b/atomic-canonical/tests/jcs_vectors.rs index 02ccc69e..3b668362 100644 --- a/atomic-canonical/tests/jcs_vectors.rs +++ b/atomic-canonical/tests/jcs_vectors.rs @@ -31,7 +31,9 @@ fn fixture(name: &str) -> Value { /// it, and require the exact bytes. fn assert_canonical(name: &str) { let f = fixture(name); - let text = f["input"].as_str().expect("fixture carries an input string"); + let text = f["input"] + .as_str() + .expect("fixture carries an input string"); let value: Value = serde_json::from_str(text).expect("fixture input is valid JSON"); let expected = f["expect"]["canonical"] .as_str() @@ -109,5 +111,8 @@ fn depth_at_the_cap_is_canonicalized() { value = Value::Array(vec![value]); } let canonical = jcs::canonicalize(&value); - assert_eq!(canonical, format!("{}null{}", "[".repeat(128), "]".repeat(128))); + assert_eq!( + canonical, + format!("{}null{}", "[".repeat(128), "]".repeat(128)) + ); }