Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
18 changes: 18 additions & 0 deletions Cargo.lock

Some generated files are not rendered by default. Learn more about how customized files appear on GitHub.

1 change: 1 addition & 0 deletions atomic-canonical/Cargo.toml
Original file line number Diff line number Diff line change
Expand Up @@ -16,4 +16,5 @@ bs58 = { workspace = true }
chrono = { workspace = true }
thiserror = { workspace = true }
data-encoding = { workspace = true }
serde_json_canonicalizer = "0.3"
atomic-identity = { workspace = true }
81 changes: 30 additions & 51 deletions atomic-canonical/src/jcs.rs
Original file line number Diff line number Diff line change
Expand Up @@ -9,63 +9,42 @@
//! hash (`hash.rs`) and the Data Integrity proof (`proof.rs`) go through
//! `canonicalize`, so the two can never drift.
//!
//! Object keys are sorted by UTF-16 code units as RFC 8785 §3.2.3 specifies
//! (not by UTF-8 bytes — the two differ once keys leave the BMP, e.g. an
//! emoji key sorts after `\u{ff61}` in UTF-8 but before it in UTF-16).
//! # The algorithm is delegated rather than written here
//!
//! Scope note: numbers are emitted via `serde_json`'s formatter, which
//! matches RFC 8785 for the integer values our vocabulary admits; the full
//! ECMAScript number-to-string algorithm is only needed if floating-point
//! payloads are ever admitted.
//! The bytes come from `serde_json_canonicalizer`. Two of the three parts of
//! RFC 8785 are straightforward to write by hand and the third is not:
//!
//! * Object keys sort by UTF-16 code units as §3.2.3 specifies, not by UTF-8
//! bytes — the two differ once a key leaves the BMP, e.g. an emoji key sorts
//! after `\u{ff61}` in UTF-8 but before it in UTF-16.
//! * Strings take the short form for the seven named escapes and lowercase
//! `\u00xx` for the rest of C0 (§3.2.2.2).
//! * **Numbers take the ECMAScript `Number::toString` algorithm** (§3.2.2.3),
//! which `serde_json`'s formatter is not. `serde_json` writes `-0.0` where the
//! algorithm writes `0`, keeps a `.0` on an integer-valued double where the
//! algorithm drops it, and crosses between decimal and exponent notation at
//! different magnitudes: `1e-6` for `0.000001`, `2.9514790517935283e+20` for
//! `295147905179352830000`. The delegate formats through `ryu_js`, which is
//! the ECMAScript variant the section names, and it serializes through an
//! explicit heap stack rather than the call stack. The cases that used to
//! diverge are pinned in `tests/vectors/`.
//!
//! Scope note: the entry point takes an already-parsed [`Value`], so faults that
//! only exist in the wire bytes cannot be decided here — a repeated object
//! member is gone before this function is called, and RFC 8785 admits an integer
//! past 2^53 by rounding it to its double. Those belong to a strict decoder at
//! the boundary where the bytes arrive; `tests/vectors/INGEST-BOUNDARY.md` names
//! the cases and what closes them.

use serde_json::Value;

/// Canonicalize a JSON value into its RFC-8785 string form.
pub fn canonicalize(value: &Value) -> String {
let mut out = String::new();
write_value(&mut out, value);
out
}

fn write_value(out: &mut String, value: &Value) {
match value {
Value::Null => out.push_str("null"),
Value::Bool(true) => out.push_str("true"),
Value::Bool(false) => out.push_str("false"),
Value::Number(n) => out.push_str(&n.to_string()),
Value::String(s) => write_json_string(out, s),
Value::Array(items) => {
out.push('[');
for (i, item) in items.iter().enumerate() {
if i > 0 {
out.push(',');
}
write_value(out, item);
}
out.push(']');
}
Value::Object(map) => {
let mut keys: Vec<&String> = map.keys().collect();
keys.sort_by(|a, b| a.encode_utf16().cmp(b.encode_utf16()));
out.push('{');
for (i, key) in keys.iter().enumerate() {
if i > 0 {
out.push(',');
}
write_json_string(out, key);
out.push(':');
write_value(out, &map[*key]);
}
out.push('}');
}
}
}

/// Emit a JSON string with standard escaping. `serde_json` produces a valid,
/// minimally-escaped JSON string literal (quotes included), which matches JCS
/// for the ASCII content in our records.
fn write_json_string(out: &mut String, s: &str) {
out.push_str(&serde_json::to_string(s).expect("string serialization is infallible"));
// Infallible for a `Value`: there is no writer to fail against, every
// member name is already a Rust `String`, and the delegate's only other
// error path is a non-finite float, which `Value` cannot hold. This mirrors
// the expectation the hand-written string helper carried before it.
serde_json_canonicalizer::to_string(value).expect("canonicalizing a Value is infallible")
}

#[cfg(test)]
Expand Down
118 changes: 118 additions & 0 deletions atomic-canonical/tests/jcs_vectors.rs
Original file line number Diff line number Diff line change
@@ -0,0 +1,118 @@
//! Canonicalization vectors, one fixture per divergence.
//!
//! Each file in `tests/vectors/` carries an input document, the divergence it
//! pins, and the canonical form RFC 8785 requires. The number cases are Appendix
//! B rows: the canonical text of a double is one string and no other, so a
//! canonicalizer that writes `1e-6` where the algorithm writes `0.000001`
//! produces a different hash for the same value, and a second implementation
//! then rejects a proof this one accepts.
//!
//! `tests/vectors/INGEST-BOUNDARY.md` lists the conformance cases this entry
//! point cannot decide, because it receives an already-parsed value rather than
//! the bytes.

use std::fs;
use std::path::{Path, PathBuf};

use atomic_canonical::jcs;
use serde_json::Value;

fn vector_dir() -> PathBuf {
Path::new(env!("CARGO_MANIFEST_DIR")).join("tests/vectors")
}

fn fixture(name: &str) -> Value {
let path = vector_dir().join(format!("{name}.json"));
let raw = fs::read_to_string(&path).unwrap_or_else(|e| panic!("read {path:?}: {e}"));
serde_json::from_str(&raw).unwrap_or_else(|e| panic!("parse {path:?}: {e}"))
}

/// Parse the fixture's input the way any caller receiving JSON does, canonicalize
/// it, and require the exact bytes.
fn assert_canonical(name: &str) {
let f = fixture(name);
let text = f["input"]
.as_str()
.expect("fixture carries an input string");
let value: Value = serde_json::from_str(text).expect("fixture input is valid JSON");
let expected = f["expect"]["canonical"]
.as_str()
.expect("fixture expects a canonical form");
assert_eq!(
jcs::canonicalize(&value),
expected,
"{name} ({}): {}",
f["vector"].as_str().unwrap_or("?"),
f["divergence"].as_str().unwrap_or("")
);
}

#[test]
fn number_decimal_below_exponent_threshold() {
assert_canonical("number-decimal-below-exponent-threshold");
}

#[test]
fn number_decimal_2pow68() {
assert_canonical("number-decimal-2pow68");
}

#[test]
fn number_decimal_999999999999999700000() {
assert_canonical("number-decimal-999999999999999700000");
}

#[test]
fn number_decimal_999999999999999900000() {
assert_canonical("number-decimal-999999999999999900000");
}

#[test]
fn number_negative_small_decimal() {
assert_canonical("number-negative-small-decimal");
}

#[test]
fn number_negative_zero() {
assert_canonical("number-negative-zero");
}

#[test]
fn number_exponent_9_999999999999997e22() {
assert_canonical("number-exponent-9.999999999999997e22");
}

#[test]
fn number_exponent_1_0000000000000001e23() {
assert_canonical("number-exponent-1.0000000000000001e23");
}

#[test]
fn number_exponent_9_999999999999997e_minus_7() {
assert_canonical("number-exponent-9.999999999999997e-7");
}

#[test]
fn number_rounded_to_its_double() {
assert_canonical("number-rounded-to-its-double");
}

/// The accept half of the depth condition: a document at 128 containers
/// canonicalizes, and the delegate walks it on the heap rather than the call
/// stack. The refusal half needs a fallible boundary -- see
/// `tests/vectors/INGEST-BOUNDARY.md`.
#[test]
fn depth_at_the_cap_is_canonicalized() {
let f = fixture("depth-at-the-cap-is-canonicalized");
assert!(f["input_generated"].as_str().is_some());

let mut value = Value::Null;
for _ in 0..128 {
value = Value::Array(vec![value]);
}
let canonical = jcs::canonicalize(&value);
assert_eq!(
canonical,
format!("{}null{}", "[".repeat(128), "]".repeat(128))
);
}
30 changes: 30 additions & 0 deletions atomic-canonical/tests/vectors/INGEST-BOUNDARY.md
Original file line number Diff line number Diff line change
@@ -0,0 +1,30 @@
# Vectors this entry point cannot decide

`jcs::canonicalize` takes an already-parsed `serde_json::Value`. Three conformance
vectors ask for a refusal that no function of that shape can give, because the
fault either vanished during the parse or is not a fault under RFC 8785 at all.
They are listed here rather than dropped, so the follow-up has its brief in-tree.

| vector | condition | what it asks for | why not here |
| --- | --- | --- | --- |
| `v0f4f2093061d303f` | duplicate member | reject `{"a":1,"a":2}` | `serde_json` keeps one of the two members while parsing, so the repeat is gone before `canonicalize` is called. Measured: the document canonicalizes to `{"a":2}`, and `{"a":2,"a":1}` to `{"a":1}` -- two wire documents, two canonical forms, and a signature over either verifies. |
| `v679f56481420e45a` | unsafe integer | reject `9007199254740993` | RFC 8785 defers number formatting to ECMAScript, which has one numeric type, so the specification *admits* the token and writes the double it rounds to. Refusing it is the RFC 7493 I-JSON profile, which is a stricter profile rather than RFC 8785 itself. Pinned as an accept in `number-rounded-to-its-double.json`. |
| `v97f5d8777e514257` | non-integer in a signed field | reject `0.7` | Same shape: RFC 8785 admits a fractional number. Refusing one is a field-level profile decision, not a canonicalization rule. |

A fourth case is half-covered. `vd94ac70c9f0d84bf` asks a canonicalizer to refuse
a document nested one container past a stated cap with a catchable error. The
accept half is pinned in `depth-at-the-cap-is-canonicalized.json`; the refusal
half needs a fallible boundary, and an infallible `canonicalize(&Value) -> String`
has nowhere to put it.

## What closes all four

A strict decoder on the raw bytes, ahead of this function: it refuses a repeated
member, caps nesting at 128 with an error rather than a stack walk, refuses a
string that is not a sequence of Unicode scalar values, and optionally applies the
I-JSON safe-integer profile. `jcs-admit` on crates.io does exactly that and then
hands canonical output to the same `serde_json_canonicalizer` this file already
uses, so adopting it adds refusals without changing a single byte of output.

The place it belongs is wherever a document arrives as bytes rather than as a
value built in-process.
Original file line number Diff line number Diff line change
@@ -0,0 +1,8 @@
{
"vector": "vd94ac70c9f0d84bf (accept half)",
"divergence": "a document nested 128 containers deep canonicalizes, which is the accept side of the depth condition",
"input_generated": "128 nested arrays around null",
"expect": {
"canonical_shape": "128 open brackets, null, 128 close brackets"
}
}
8 changes: 8 additions & 0 deletions atomic-canonical/tests/vectors/number-decimal-2pow68.json
Original file line number Diff line number Diff line change
@@ -0,0 +1,8 @@
{
"vector": "v362600d69975d14c",
"divergence": "decimal-to-exponent threshold: a magnitude below 1e21 stays decimal",
"input": "{\"n\":295147905179352830000}",
"expect": {
"canonical": "{\"n\":295147905179352830000}"
}
}
Original file line number Diff line number Diff line change
@@ -0,0 +1,8 @@
{
"vector": "vaa33c34f7bd1058b",
"divergence": "decimal notation retained just below the 1e21 threshold",
"input": "{\"n\":999999999999999700000}",
"expect": {
"canonical": "{\"n\":999999999999999700000}"
}
}
Original file line number Diff line number Diff line change
@@ -0,0 +1,8 @@
{
"vector": "vb177ef3b3a945a72",
"divergence": "decimal notation retained for the last double below 1e21",
"input": "{\"n\":999999999999999900000}",
"expect": {
"canonical": "{\"n\":999999999999999900000}"
}
}
Original file line number Diff line number Diff line change
@@ -0,0 +1,8 @@
{
"vector": "vaf1b67f038ffff84",
"divergence": "decimal-to-exponent threshold: 1e-6 is written in decimal notation",
"input": "{\"n\":0.000001}",
"expect": {
"canonical": "{\"n\":0.000001}"
}
}
Original file line number Diff line number Diff line change
@@ -0,0 +1,8 @@
{
"vector": "vb7dd5fb8dcb6e345",
"divergence": "seventeen significant digits kept where sixteen would not round-trip",
"input": "{\"n\":1.0000000000000001e+23}",
"expect": {
"canonical": "{\"n\":1.0000000000000001e+23}"
}
}
Original file line number Diff line number Diff line change
@@ -0,0 +1,8 @@
{
"vector": "vf2f6991c55096b1a",
"divergence": "shortest round-tripping digits below the 1e-6 threshold",
"input": "{\"n\":9.999999999999997e-7}",
"expect": {
"canonical": "{\"n\":9.999999999999997e-7}"
}
}
Original file line number Diff line number Diff line change
@@ -0,0 +1,8 @@
{
"vector": "v843566cf72cd33c3",
"divergence": "shortest round-tripping digits in exponent notation",
"input": "{\"n\":9.999999999999997e+22}",
"expect": {
"canonical": "{\"n\":9.999999999999997e+22}"
}
}
Original file line number Diff line number Diff line change
@@ -0,0 +1,8 @@
{
"vector": "vd2659c36ece7eb47",
"divergence": "a negative magnitude above 1e-6 stays decimal",
"input": "{\"n\":-0.0000033333333333333333}",
"expect": {
"canonical": "{\"n\":-0.0000033333333333333333}"
}
}
8 changes: 8 additions & 0 deletions atomic-canonical/tests/vectors/number-negative-zero.json
Original file line number Diff line number Diff line change
@@ -0,0 +1,8 @@
{
"vector": "v9a8b364b8bc121de",
"divergence": "negative zero is written as 0, and -0.0 is not valid ECMAScript output at all",
"input": "{\"n\":-0.0}",
"expect": {
"canonical": "{\"n\":0}"
}
}
Original file line number Diff line number Diff line change
@@ -0,0 +1,8 @@
{
"vector": "v679f56481420e45a",
"divergence": "RFC 8785 treats every number as a double, so 2^53+1 canonicalizes to the double it rounds to",
"input": "{\"n\":9007199254740993}",
"expect": {
"canonical": "{\"n\":9007199254740992}"
}
}
Loading