From 9214da8b531e72fa829aa762056982fc8a759a66 Mon Sep 17 00:00:00 2001 From: Sankalp Gilda Date: Sun, 20 Sep 2026 01:31:15 -0400 Subject: [PATCH 01/22] feat(canonical): admit JSON on the bytes before it is hashed or signed jcs::canonicalize takes a serde_json::Value, and four conformance cases cannot be decided from one. A repeated object member is gone before canonicalize is reached, because the parse kept the last of the two. A document nested past a bound has already been built by the time anything could decline to build it. RFC 8785 section 3.2.2.3 defers number formatting to ECMAScript, which has one numeric type, so the specification admits an integer past 2^53 and a fractional number and writes the double each rounds to; refusing either is the RFC 7493 profile rather than a canonicalization rule. Add jcs::admit_document(&[u8]) -> Result. It runs jcs-admit over the raw bytes under RFC 8785 plus the RFC 7493 profile tightened to integers only, then parses the same bytes. The value comes from the bytes the document arrived in rather than from the admission's canonical output, so an accepted document keeps its value, its content hash and its proof, and refusal is the only new behaviour. CanonicalError::Admission carries the fault itself rather than a rendered string, so a caller can tell a repeated member from a depth bound. Two call sites move onto it: decode_from_transport, which receives the Atomic-Delegation header, and load_for_delegate, which reads documents back out of the identity store. maxChanges is the only numeric field the vocabulary carries and it is a count, so the integers-only tightening costs nothing. Nine tests in atomic-canonical/tests/ingest_boundary.rs. Four are the cases above, copied byte for byte under tests/vectors/ingest/ with their vector ids, manifest paths, suite commit and digests recorded in PROVENANCE.md, each asserting the refusal variant rather than a message. Five are accept controls over a minted certificate: admitted in both the compact and the indented serialization, still round-tripping through encode_for_transport and verifying, and still found in a real identity store. --- Cargo.lock | 94 ++++-- Cargo.toml | 6 + atomic-canonical/Cargo.toml | 6 + atomic-canonical/src/delegation.rs | 24 +- atomic-canonical/src/error.rs | 9 + atomic-canonical/src/jcs.rs | 58 ++++ atomic-canonical/tests/ingest_boundary.rs | 248 ++++++++++++++ .../tests/vectors/ingest/PROVENANCE.md | 45 +++ .../ingest/records/v0f4f2093061d303f.jsonl | 1 + .../ingest/records/v97f5d8777e514257.jsonl | 1 + .../ingest/statements/v679f56481420e45a.json | 44 +++ .../ingest/statements/vd94ac70c9f0d84bf.json | 303 ++++++++++++++++++ 12 files changed, 804 insertions(+), 35 deletions(-) create mode 100644 atomic-canonical/tests/ingest_boundary.rs create mode 100644 atomic-canonical/tests/vectors/ingest/PROVENANCE.md create mode 100644 atomic-canonical/tests/vectors/ingest/records/v0f4f2093061d303f.jsonl create mode 100644 atomic-canonical/tests/vectors/ingest/records/v97f5d8777e514257.jsonl create mode 100644 atomic-canonical/tests/vectors/ingest/statements/v679f56481420e45a.json create mode 100644 atomic-canonical/tests/vectors/ingest/statements/vd94ac70c9f0d84bf.json diff --git a/Cargo.lock b/Cargo.lock index a4bd3044..05134198 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -167,8 +167,10 @@ dependencies = [ "bs58", "chrono", "data-encoding", + "jcs-admit", "serde", "serde_json", + "tempfile", "thiserror 1.0.69", ] @@ -591,7 +593,7 @@ dependencies = [ "heck", "proc-macro2", "quote", - "syn", + "syn 2.0.118", ] [[package]] @@ -775,7 +777,7 @@ checksum = "f46882e17999c6cc590af592290432be3bce0428cb0d5f8b6715e4dc7b383eb3" dependencies = [ "proc-macro2", "quote", - "syn", + "syn 2.0.118", ] [[package]] @@ -803,7 +805,7 @@ dependencies = [ "defmt-parser", "proc-macro2", "quote", - "syn", + "syn 2.0.118", ] [[package]] @@ -877,7 +879,7 @@ checksum = "1ac70aa55017e108007fbaf5aa0f54b021c98f92ff8af59d42eda9da96e3dd4f" dependencies = [ "proc-macro2", "quote", - "syn", + "syn 2.0.118", ] [[package]] @@ -1135,7 +1137,7 @@ checksum = "e835b70203e41293343137df5c0664546da5745f82ec9b84d40be8336958447b" dependencies = [ "proc-macro2", "quote", - "syn", + "syn 2.0.118", ] [[package]] @@ -1647,6 +1649,16 @@ version = "1.0.18" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "8f42a60cbdf9a97f5d2305f08a87dc4e09308d1276d28c869c684d7777685682" +[[package]] +name = "jcs-admit" +version = "0.1.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "e5e809813f7ed9d15536d3038657aee1489e5f4f4b2df5b58519007aebf8f9a9" +dependencies = [ + "serde", + "serde_json_canonicalizer", +] + [[package]] name = "jiff" version = "0.2.32" @@ -1669,7 +1681,7 @@ checksum = "d0879bd39df99c4c5e2c6615ccc026391a423dde10532c573e6086eb94a802cc" dependencies = [ "proc-macro2", "quote", - "syn", + "syn 2.0.118", ] [[package]] @@ -1916,7 +1928,7 @@ checksum = "a948666b637a0f465e8564c73e89d4dde00d72d4d473cc972f390fc3dcee7d9c" dependencies = [ "proc-macro2", "quote", - "syn", + "syn 2.0.118", ] [[package]] @@ -2084,7 +2096,7 @@ checksum = "a9a28b8493dd664c8b171dd944da82d933f7d456b829bfb236738e1fe06c5ba4" dependencies = [ "proc-macro2", "quote", - "syn", + "syn 2.0.118", ] [[package]] @@ -2508,6 +2520,12 @@ version = "1.0.23" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "9774ba4a74de5f7b1c1451ed6cd5285a32eddb5cccb8cc655a4e50009e06477f" +[[package]] +name = "ryu-js" +version = "1.0.3" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "04d056b875a9d2e6cb9a61d127afee9ac5999b9f87bcb32079d1318e505be714" + [[package]] name = "same-file" version = "1.0.6" @@ -2563,9 +2581,9 @@ checksum = "8a7852d02fc848982e0c167ef163aaff9cd91dc640ba85e263cb1ce46fae51cd" [[package]] name = "serde" -version = "1.0.228" +version = "1.0.229" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "9a8e94ea7f378bd32cbbd37198a4a91436180c5bb472411e48b5ec2e2124ae9e" +checksum = "4148590afebada386688f18773da617792bf2ef03ffc1e4cbd2b1d45b023e0ba" dependencies = [ "serde_core", "serde_derive", @@ -2597,22 +2615,22 @@ dependencies = [ [[package]] name = "serde_core" -version = "1.0.228" +version = "1.0.229" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "41d385c7d4ca58e59fc732af25c3983b67ac852c1a25000afe1175de458b67ad" +checksum = "67dca2c9c51e58a4791a4b1ed58308b39c64224d349a935ab5039aa360942a48" dependencies = [ "serde_derive", ] [[package]] name = "serde_derive" -version = "1.0.228" +version = "1.0.229" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "d540f220d3187173da220f885ab66608367b6574e925011a9353e4badda91d79" +checksum = "e7a5d71263a5a7d47b41f6b3f06ba276f10cc18b0931f1799f710578e2309348" dependencies = [ "proc-macro2", "quote", - "syn", + "syn 3.0.6", ] [[package]] @@ -2629,6 +2647,17 @@ dependencies = [ "zmij", ] +[[package]] +name = "serde_json_canonicalizer" +version = "0.3.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "fe52319a927259afbfa5180c5157cd8167edfd3e8c254f9558c7fef44c5649f2" +dependencies = [ + "ryu-js", + "serde", + "serde_json", +] + [[package]] name = "serde_spanned" version = "0.6.9" @@ -2672,7 +2701,7 @@ checksum = "94e153fc76e1c6a068703d6d29c508a0b15c061c4b7e43da59cc097bc342673c" dependencies = [ "proc-macro2", "quote", - "syn", + "syn 2.0.118", ] [[package]] @@ -2805,6 +2834,17 @@ dependencies = [ "unicode-ident", ] +[[package]] +name = "syn" +version = "3.0.6" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "8593e8e72159ed2257d083c7a454a85cbf854f37a0966d8d483aff8c8a3ebcee" +dependencies = [ + "proc-macro2", + "quote", + "unicode-ident", +] + [[package]] name = "sync_wrapper" version = "1.0.2" @@ -2822,7 +2862,7 @@ checksum = "728a70f3dbaf5bab7f0c4b1ac8d7ae5ea60a4b5549c8a5914361c99147a709d2" dependencies = [ "proc-macro2", "quote", - "syn", + "syn 2.0.118", ] [[package]] @@ -2928,7 +2968,7 @@ checksum = "4fee6c4efc90059e10f81e6d42c60a18f76588c3d74cb83a0b242a2b6c7504c1" dependencies = [ "proc-macro2", "quote", - "syn", + "syn 2.0.118", ] [[package]] @@ -2939,7 +2979,7 @@ checksum = "ebc4ee7f67670e9b64d05fa4253e753e016c6c95ff35b89b7941d6b856dec1d5" dependencies = [ "proc-macro2", "quote", - "syn", + "syn 2.0.118", ] [[package]] @@ -2992,7 +3032,7 @@ checksum = "385a6cb71ab9ab790c5fe8d67f1645e6c450a7ce006a33de03daa956cf70a496" dependencies = [ "proc-macro2", "quote", - "syn", + "syn 2.0.118", ] [[package]] @@ -3424,7 +3464,7 @@ dependencies = [ "bumpalo", "proc-macro2", "quote", - "syn", + "syn 2.0.118", "wasm-bindgen-shared", ] @@ -3581,7 +3621,7 @@ checksum = "053e2e040ab57b9dc951b72c264860db7eb3b0200ba345b4e4c3b14f67855ddf" dependencies = [ "proc-macro2", "quote", - "syn", + "syn 2.0.118", ] [[package]] @@ -3592,7 +3632,7 @@ checksum = "3f316c4a2570ba26bbec722032c4099d8c8bc095efccdc15688708623367e358" dependencies = [ "proc-macro2", "quote", - "syn", + "syn 2.0.118", ] [[package]] @@ -3862,7 +3902,7 @@ checksum = "de844c262c8848816172cef550288e7dc6c7b7814b4ee56b3e1553f275f1858e" dependencies = [ "proc-macro2", "quote", - "syn", + "syn 2.0.118", "synstructure", ] @@ -3883,7 +3923,7 @@ checksum = "e2e817b7b52d0c7358d3246da9d69935ebb18116b2b102b4230dac079b4862f5" dependencies = [ "proc-macro2", "quote", - "syn", + "syn 2.0.118", ] [[package]] @@ -3903,7 +3943,7 @@ checksum = "11532158c46691caf0f2593ea8358fed6bbf68a0315e80aae9bd41fbade684a1" dependencies = [ "proc-macro2", "quote", - "syn", + "syn 2.0.118", "synstructure", ] @@ -3943,7 +3983,7 @@ checksum = "625dc425cab0dca6dc3c3319506e6593dcb08a9f387ea3b284dbd52a92c40555" dependencies = [ "proc-macro2", "quote", - "syn", + "syn 2.0.118", ] [[package]] diff --git a/Cargo.toml b/Cargo.toml index bcc80d15..fc684a06 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -39,6 +39,12 @@ atomic-teams = { path = "atomic-teams" } # Serialization serde = { version = "1.0", features = ["derive"] } serde_json = "1.0" +# Admission control for JSON that will be hashed, signed or verified. Runs on +# the raw bytes, ahead of the parse, so the faults a parsed value can no longer +# report -- a repeated member, nesting past the depth cap, a string that is not +# a sequence of Unicode scalar values, a number outside the profile -- are +# refused with a named error rather than collapsed or rounded. +jcs-admit = "0.1" postcard = { version = "1.0", features = ["alloc"] } toml = "0.8" semver = "1" diff --git a/atomic-canonical/Cargo.toml b/atomic-canonical/Cargo.toml index 58ceac78..f5004d9f 100644 --- a/atomic-canonical/Cargo.toml +++ b/atomic-canonical/Cargo.toml @@ -11,9 +11,15 @@ rust-version.workspace = true [dependencies] serde = { workspace = true } serde_json = { workspace = true } +jcs-admit = { workspace = true } blake3 = { workspace = true } bs58 = { workspace = true } chrono = { workspace = true } thiserror = { workspace = true } data-encoding = { workspace = true } atomic-identity = { workspace = true } + +[dev-dependencies] +# The ingest-boundary test drives the real identity store, because the bytes +# `load_for_delegate` admits are the bytes a store holds. +tempfile = { workspace = true } diff --git a/atomic-canonical/src/delegation.rs b/atomic-canonical/src/delegation.rs index d7fe2f38..31bc9e9d 100644 --- a/atomic-canonical/src/delegation.rs +++ b/atomic-canonical/src/delegation.rs @@ -296,10 +296,16 @@ pub fn encode_for_transport(document: &Value) -> String { /// Decode a certificate presented in a request header. /// -/// Checks the size cap first, then base64, then JSON. Does **not** verify — +/// Checks the size cap first, then base64, then admission. Does **not** verify — /// [`verify`] against the delegator's registered key is a separate, mandatory /// step, and keeping them apart means no call site can accidentally treat a /// well-formed certificate as a trusted one. +/// +/// Admission runs through [`crate::jcs::admit_document`] rather than a plain +/// parse, because this is the one place a caller-supplied document reaches the +/// shared canonicalizer. A certificate whose `delegateKey` is repeated names one +/// key to a reader and a different one to the signature, and the parse discarded +/// the repeat before anything could refuse it. pub fn decode_from_transport(encoded: &str) -> Result { if encoded.len() > MAX_ENCODED_DELEGATION { return Err(CanonicalError::Proof(format!( @@ -312,8 +318,7 @@ pub fn decode_from_transport(encoded: &str) -> Result { .decode(encoded.trim().as_bytes()) .map_err(|e| CanonicalError::Proof(format!("delegation is not valid base64url: {e}")))?; - serde_json::from_slice(&bytes) - .map_err(|e| CanonicalError::Proof(format!("delegation is not valid JSON: {e}"))) + jcs::admit_document(&bytes) } /// A stable fingerprint of an encoded certificate, for caching a verified @@ -506,10 +511,11 @@ impl StoredDelegation { /// Every verified certificate in `store` naming `delegate` as its subject, /// newest first. /// -/// Certificates that fail to parse or verify are **skipped**, not returned as -/// errors. This is the one place a corrupt or foreign file in the store could -/// otherwise take down every agent operation, and a certificate that does not -/// verify has no authority to convey in any case. Each skip is logged at warn. +/// Certificates that fail admission or verification are **skipped**, not +/// returned as errors. This is the one place a corrupt or foreign file in the +/// store could otherwise take down every agent operation, and a certificate that +/// does not verify has no authority to convey in any case. Each skip is logged +/// at warn. /// /// Verification is self-contained — it uses the delegator key the certificate /// carries — so this works on a machine that holds only the agent's key. @@ -524,7 +530,9 @@ pub fn load_for_delegate( let mut out = Vec::new(); for (id, raw) in stored { - let Ok(value) = serde_json::from_str::(&raw) else { + // Admission, not a plain parse: a stored certificate is a document this + // machine received from somewhere else, and it is about to be verified. + let Ok(value) = jcs::admit_document(raw.as_bytes()) else { continue; }; // Cheap discriminator before the Ed25519 verify: most certificates in diff --git a/atomic-canonical/src/error.rs b/atomic-canonical/src/error.rs index dff93241..9d7c7dfc 100644 --- a/atomic-canonical/src/error.rs +++ b/atomic-canonical/src/error.rs @@ -21,6 +21,15 @@ pub enum CanonicalError { #[error("identity error: {0}")] Identity(#[from] atomic_identity::IdentityError), + + /// A document arriving as bytes was refused before it became a value. + /// + /// Carries the admission fault itself rather than a rendered string, so a + /// caller can branch on it: a repeated member means two readings of one + /// document and a depth refusal is a bound on the work, and the two want + /// different handling. + #[error("document refused at the ingest boundary: {0}")] + Admission(#[from] jcs_admit::Error), } pub type Result = std::result::Result; diff --git a/atomic-canonical/src/jcs.rs b/atomic-canonical/src/jcs.rs index b1ec4d09..2930e56e 100644 --- a/atomic-canonical/src/jcs.rs +++ b/atomic-canonical/src/jcs.rs @@ -17,9 +17,67 @@ //! matches RFC 8785 for the integer values our vocabulary admits; the full //! ECMAScript number-to-string algorithm is only needed if floating-point //! payloads are ever admitted. +//! +//! # Admission is a separate step, and it happens earlier +//! +//! `canonicalize` takes a value someone has already parsed. Several faults that +//! matter to a verifier are gone by then: a repeated member survives only until +//! the parse collapses it to last-wins, and a document nested past any bound has +//! already been built by the time anything could decline to build it. +//! [`admit_document`] is the entry point for a document that arrives as *bytes* +//! — off a request header, off disk, off stdin — and it decides those questions +//! where they are still answerable. Every path in the workspace that receives a +//! certificate as bytes and later hashes, signs or verifies it goes through this +//! function, so the decision happens before the value exists. use serde_json::Value; +use crate::error::Result; + +/// What a document arriving as bytes is admitted under. +/// +/// RFC 8785 as written, plus the RFC 7493 I-JSON profile, plus integers only. +/// The profile is not decoration: +/// +/// - **Safe integers.** RFC 8785 section 3.2.2.3 defers number formatting to +/// ECMAScript, which has one numeric type, so a conforming implementation +/// canonicalizes `9007199254740993` to `9007199254740992` — a different +/// integer, with no error. A signed field whose value a re-checker is entitled +/// to rewrite is a field two parties can disagree about. +/// - **Integers only.** Nothing in the vocabulary needs a fractional number: +/// the one numeric field a certificate carries is `maxChanges`, a count. Two +/// implementations that never format a float can never disagree about one. +const ADMISSION: jcs_admit::Options = jcs_admit::Options::ijson().integers_only(true); + +/// Admit a JSON document that arrived as bytes, and return the value it denotes. +/// +/// The refusals are the point. A repeated member name gives one document two +/// readings: two parties take the same bytes for two different documents and a +/// signature over either reading verifies, and by the time a `serde_json::Value` +/// exists the repeat is gone. Nesting past 128 containers is refused here rather +/// than walked, because the depth bound this crate's hashing and proof paths +/// rely on today is `serde_json`'s incidental parser default rather than one +/// either of them asserts. A number outside the profile above is refused rather +/// than rounded. +/// +/// The accepted document is parsed from the bytes it arrived in, not from the +/// admission's canonical output, so nothing about an accepted document changes: +/// the value, its content hash and its proof are exactly what they were. This +/// function only adds refusals. +/// +/// # Errors +/// +/// [`crate::CanonicalError::Admission`] when the bytes are refused, carrying the +/// named fault and its byte offset; [`crate::CanonicalError::Proof`] if the +/// admitted bytes then fail to deserialize, which is a disagreement between the +/// admission layer and `serde_json` rather than a statement about the input. +pub fn admit_document(bytes: &[u8]) -> Result { + jcs_admit::admit_with(bytes, &ADMISSION)?; + serde_json::from_slice(bytes).map_err(|e| { + crate::error::CanonicalError::Proof(format!("admitted document does not deserialize: {e}")) + }) +} + /// Canonicalize a JSON value into its RFC-8785 string form. pub fn canonicalize(value: &Value) -> String { let mut out = String::new(); diff --git a/atomic-canonical/tests/ingest_boundary.rs b/atomic-canonical/tests/ingest_boundary.rs new file mode 100644 index 00000000..adc5e307 --- /dev/null +++ b/atomic-canonical/tests/ingest_boundary.rs @@ -0,0 +1,248 @@ +//! The four conformance cases a value-shaped entry point cannot decide. +//! +//! `jcs::canonicalize` takes a `serde_json::Value`, and by the time one exists a +//! repeated member has collapsed to last-wins and a document nested past any +//! bound has already been built. `jcs::admit_document` sits on the bytes +//! instead, which is where those questions are still answerable. +//! +//! Every reject fixture in `tests/vectors/ingest/` is a byte-for-byte copy of a +//! conformance document; `tests/vectors/ingest/PROVENANCE.md` names the vector +//! each came from, the path the suite's manifest gives it, and the suite commit, +//! and it says which two are a vector's record sidecar rather than its statement +//! because the sidecar is the half carrying the fault. Each test below +//! asserts the refusal variant rather than a message, because the variant is what +//! a verifier branches on: a repeated member means two readings of one document +//! and a depth refusal is a bound on the work. +//! +//! The accept controls matter as much. A gate that refuses everything scores the +//! same as a gate that refuses nothing, so each boundary in this crate is +//! exercised with a certificate it must still take, as well as with one it must +//! now refuse. + +use std::fs; +use std::path::{Path, PathBuf}; + +use atomic_canonical::delegation; +use atomic_canonical::jcs; +use atomic_canonical::CanonicalError; +use atomic_identity::delegation::{Delegation, DelegationScope}; +use atomic_identity::identity::{Identity, IdentityType}; +use atomic_identity::keypair::KeyPair; +use atomic_identity::IdentityStore; +use jcs_admit::Error as Admission; +use serde_json::Value; + +fn vector(rel: &str) -> Vec { + let path: PathBuf = Path::new(env!("CARGO_MANIFEST_DIR")) + .join("tests/vectors/ingest") + .join(rel); + fs::read(&path).unwrap_or_else(|e| panic!("read {path:?}: {e}")) +} + +fn admission_error(bytes: &[u8]) -> Admission { + match jcs::admit_document(bytes) { + Err(CanonicalError::Admission(e)) => e, + Err(other) => panic!("expected an admission refusal, got {other}"), + Ok(_) => panic!("expected a refusal, the document was admitted"), + } +} + +// --------------------------------------------------------------------------- +// The four cases +// --------------------------------------------------------------------------- + +/// `v0f4f2093061d303f` (`aia-c-5`). One wire document, two readings: a +/// first-wins reader shows `read_file` while the hash commits to +/// `delete_repository`. A signature over either reading verifies, so the fault +/// has to be refused before a reading is chosen. +#[test] +fn duplicate_member_is_refused() { + let bytes = vector("records/v0f4f2093061d303f.jsonl"); + assert!( + matches!(admission_error(&bytes), Admission::DuplicateMember { ref name, .. } if name == "toolName"), + "v0f4f2093061d303f must be refused as a duplicate member" + ); + // And the reason a value-shaped check cannot do it: the repeat is gone. + let parsed: Value = serde_json::from_slice(&bytes).expect("the bytes do parse"); + assert_eq!(parsed["toolName"], "delete_repository"); +} + +/// `vd94ac70c9f0d84bf` (`aia-c-12`). One container past the 128 cap, refused +/// with a catchable error naming the cap rather than by recursing. The cap this +/// crate's hashing and proof paths rely on today is `serde_json`'s incidental +/// parser default, which is a bound neither of them asserts. +#[test] +fn nesting_one_past_the_cap_is_refused() { + let bytes = vector("statements/vd94ac70c9f0d84bf.json"); + assert!( + matches!( + admission_error(&bytes), + Admission::TooDeep { limit: 128, .. } + ), + "vd94ac70c9f0d84bf must be refused at the depth cap" + ); +} + +/// `v679f56481420e45a` (`aia-c-14`). 2^53 + 1 in `durationMs`. RFC 8785 defers +/// number formatting to ECMAScript, which has one numeric type, so a conforming +/// re-checker canonicalizes this to `9007199254740992` — a different integer, +/// with no error. A signed field a re-checker may rewrite is a field two parties +/// can disagree about, which is what the RFC 7493 profile removes. +#[test] +fn an_integer_past_2pow53_is_refused() { + let bytes = vector("statements/v679f56481420e45a.json"); + assert!( + matches!(admission_error(&bytes), Admission::UnsafeInteger { ref token } if token == "9007199254740993"), + "v679f56481420e45a must be refused as an unsafe integer" + ); +} + +/// `v97f5d8777e514257` (`aia-c-9`). A fractional `durationMs` in a signed +/// record. RFC 8785 admits it and writes the double; two implementations that +/// never format a float cannot disagree about one, so the profile refuses it and +/// leaves fractional values to the content-digest form. +#[test] +fn a_non_integer_in_a_signed_field_is_refused() { + let bytes = vector("records/v97f5d8777e514257.jsonl"); + assert!( + matches!(admission_error(&bytes), Admission::NonIntegerNumber { ref token } if token == "412.5"), + "v97f5d8777e514257 must be refused as a non-integer" + ); +} + +// --------------------------------------------------------------------------- +// Accept controls, one per call site the change touches +// --------------------------------------------------------------------------- + +struct Pair { + human: Identity, + human_key: KeyPair, + agent: Identity, +} + +fn pair() -> Pair { + let human_key = KeyPair::generate(); + let human = Identity::new("alice", &human_key); + let agent_key = KeyPair::generate(); + let agent = Identity::builder("alice+claude") + .identity_type(IdentityType::Agent) + .public_key(agent_key.public.clone()) + .delegated_by(human.id) + .build() + .expect("agent identity"); + Pair { + human, + human_key, + agent, + } +} + +fn certificate(p: &Pair) -> Value { + let scope = DelegationScope::builder() + .permission(atomic_identity::delegation::DelegationPermission::Record) + .project("acme/*") + .max_changes(64) + .build(); + let terms = Delegation::new(&p.human, &p.agent, scope); + delegation::mint(&p.human, &p.human_key, &terms) +} + +/// A real certificate is admitted in both serializations that reach the +/// boundary: the compact bytes transport carries and the indented bytes the +/// store holds. `maxChanges` is the one numeric field in the vocabulary, and an +/// honest count passes the integers-only profile. +#[test] +fn a_real_certificate_is_admitted_in_both_serializations() { + let doc = certificate(&pair()); + for rendered in [ + serde_json::to_vec(&doc).expect("compact"), + serde_json::to_vec_pretty(&doc).expect("indented"), + ] { + let admitted = jcs::admit_document(&rendered).expect("a minted certificate is admissible"); + assert_eq!( + admitted, doc, + "admission must not alter an accepted document" + ); + } +} + +/// `delegation::decode_from_transport` — the `Atomic-Delegation` request header. +/// The accept half: a certificate round-trips and still verifies. +#[test] +fn decode_from_transport_still_takes_a_real_certificate() { + let p = pair(); + let doc = certificate(&p); + let decoded = delegation::decode_from_transport(&delegation::encode_for_transport(&doc)) + .expect("a minted certificate decodes"); + assert_eq!(decoded, doc); + delegation::verify(&decoded, &p.human.public_key).expect("and verifies"); +} + +/// The same boundary, refusing. A repeated `delegateKey` names one key to a +/// reader and a different one to the signature; before this change the parse +/// discarded the repeat and the certificate went on to be verified. +#[test] +fn decode_from_transport_refuses_a_repeated_member() { + let repeated = + br#"{"@type":"AgentDelegation","delegateKey":"did:key:zFirst","delegateKey":"did:key:zSecond"}"#; + let encoded = data_encoding::BASE64URL_NOPAD.encode(repeated); + match delegation::decode_from_transport(&encoded) { + Err(CanonicalError::Admission(Admission::DuplicateMember { name, .. })) => { + assert_eq!(name, "delegateKey"); + } + other => panic!("expected a duplicate-member refusal, got {other:?}"), + } +} + +/// `delegation::load_for_delegate` — certificates read back off the identity +/// store, which is a document this machine received from somewhere else. The +/// accept half: a stored certificate is still found. +#[test] +fn load_for_delegate_still_finds_a_stored_certificate() { + let root = tempfile::tempdir().expect("temp store"); + let store = IdentityStore::open(root.path()).expect("open store"); + let p = pair(); + let doc = certificate(&p); + let parsed = delegation::parse(&doc).expect("parse the minted certificate"); + store + .save_delegation( + &parsed.id.to_base32(), + &serde_json::to_string_pretty(&doc).expect("indented"), + ) + .expect("save"); + + let found = delegation::load_for_delegate(&store, &p.agent).expect("load"); + assert_eq!(found.len(), 1); + assert_eq!(found[0].delegation.delegate_name, "alice+claude"); +} + +/// The same boundary, refusing. A stored document carrying a repeated member is +/// skipped rather than verified, and the skip does not take the other +/// certificates with it. +#[test] +fn load_for_delegate_skips_a_stored_document_with_a_repeated_member() { + let root = tempfile::tempdir().expect("temp store"); + let store = IdentityStore::open(root.path()).expect("open store"); + let p = pair(); + let doc = certificate(&p); + let parsed = delegation::parse(&doc).expect("parse the minted certificate"); + store + .save_delegation( + &parsed.id.to_base32(), + &serde_json::to_string(&doc).expect("compact"), + ) + .expect("save the good one"); + + let good = serde_json::to_string(&doc).expect("compact"); + let repeated = good.replacen('{', r#"{"delegateKey":"did:key:zSecond","#, 1); + store + .save_delegation("AAAABBBBCCCCDDDD", &repeated) + .expect("save the one with the repeat"); + + let found = delegation::load_for_delegate(&store, &p.agent).expect("load"); + assert_eq!( + found.len(), + 1, + "the admissible certificate survives and the repeated member does not" + ); +} diff --git a/atomic-canonical/tests/vectors/ingest/PROVENANCE.md b/atomic-canonical/tests/vectors/ingest/PROVENANCE.md new file mode 100644 index 00000000..ebf8304e --- /dev/null +++ b/atomic-canonical/tests/vectors/ingest/PROVENANCE.md @@ -0,0 +1,45 @@ +# Ingest-boundary fixtures + +Four documents from the conformance suite, copied byte for byte, one per +conformance case. Each carries a fault that a function taking an already-parsed +value cannot report, so each is a test of `jcs::admit_document` rather than of +`jcs::canonicalize`. + +Source: the `ai-agent-action` conformance suite, commit `cf3d20e`, at +. Paths below are the ones +`MANIFEST.json` names, and the layout here mirrors them. Nothing was reformatted: +a repeated member and a fractional number survive a pretty-printer, but the point +of a byte fixture is that it does not have to. + +A vector's id is `v` plus sixteen hex digits of SHA-256 over the vector's own +bytes: the statement alone, or the statement and a NUL and the record sidecar +where one ships. Checked against all 53 vectors in the manifest at that commit, +and it holds for all 53. So `statements/v679f56481420e45a.json` hashes to +`679f56481420e45a...` because that vector is its statement, while +`v0f4f2093061d303f` and `v97f5d8777e514257` are each a statement and a record +together and neither half hashes to the id on its own. + +For those two the copied file is the record, because the record is the half that +carries the fault: the statements are conformant documents whose declared digest +commits to a record line that is not. So the four files here are four faults, not +four whole vectors, and the id in each row names the vector the fault came from. + +| file | vector | manifest path | condition | expected refusal | +| --- | --- | --- | --- | --- | +| `records/v0f4f2093061d303f.jsonl` | `v0f4f2093061d303f` | `records/v0f4f2093061d303f.jsonl` | `aia-c-5` duplicate member | `Error::DuplicateMember { name: "toolName", offset: 130 }` | +| `records/v97f5d8777e514257.jsonl` | `v97f5d8777e514257` | `records/v97f5d8777e514257.jsonl` | `aia-c-9` non-integer in a signed field | `Error::NonIntegerNumber { token: "412.5" }` | +| `statements/v679f56481420e45a.json` | `v679f56481420e45a` | `statements/v679f56481420e45a.json` | `aia-c-14` unsafe integer | `Error::UnsafeInteger { token: "9007199254740993" }` | +| `statements/vd94ac70c9f0d84bf.json` | `vd94ac70c9f0d84bf` | `statements/vd94ac70c9f0d84bf.json` | `aia-c-12` depth exceeded | `Error::TooDeep { limit: 128, offset: 17860 }` | + +SHA-256 of each file as committed: + +``` +9d91bad0d5ed10edb29785fc66f6ca406895890f5828d96e1cec0115c8a97cce records/v0f4f2093061d303f.jsonl +52251e852eab4dc9eeb26ce0f9ccf13e3852c4ff2bfbb5e1b7380d34c2ef59c0 records/v97f5d8777e514257.jsonl +679f56481420e45a6196f2be61f29d51cc76b011e04bd8df8d6af6064c53511b statements/v679f56481420e45a.json +d94ac70c9f0d84bf1fc287376d1e2416785ce51df443a97a2d13765f5867883e statements/vd94ac70c9f0d84bf.json +``` + +Two of the four are refused by RFC 8785 alone; the other two need the RFC 7493 +profile `jcs::admit_document` applies. `tests/ingest_boundary.rs` says which is +which, and asserts the variant rather than a message. diff --git a/atomic-canonical/tests/vectors/ingest/records/v0f4f2093061d303f.jsonl b/atomic-canonical/tests/vectors/ingest/records/v0f4f2093061d303f.jsonl new file mode 100644 index 00000000..6829c0c6 --- /dev/null +++ b/atomic-canonical/tests/vectors/ingest/records/v0f4f2093061d303f.jsonl @@ -0,0 +1 @@ +{"durationMs":412,"id":"r1","previousHash":"genesis","success":true,"timestamp":"2026-08-18T14:33:41.882Z","toolName":"read_file","toolName":"delete_repository","type":"tool_call"} diff --git a/atomic-canonical/tests/vectors/ingest/records/v97f5d8777e514257.jsonl b/atomic-canonical/tests/vectors/ingest/records/v97f5d8777e514257.jsonl new file mode 100644 index 00000000..66c1f5e1 --- /dev/null +++ b/atomic-canonical/tests/vectors/ingest/records/v97f5d8777e514257.jsonl @@ -0,0 +1 @@ +{"durationMs":412.5,"extensions":{"10":"ten","2":"two","aa":"first","zz":"last"},"id":"r1","previousHash":"genesis","success":true,"timestamp":"2026-08-18T14:33:41.882Z","toolName":"create_pull_request","type":"tool_call"} diff --git a/atomic-canonical/tests/vectors/ingest/statements/v679f56481420e45a.json b/atomic-canonical/tests/vectors/ingest/statements/v679f56481420e45a.json new file mode 100644 index 00000000..5b512563 --- /dev/null +++ b/atomic-canonical/tests/vectors/ingest/statements/v679f56481420e45a.json @@ -0,0 +1,44 @@ +{ + "_type": "https://in-toto.io/Statement/v1", + "predicate": { + "action": { + "durationMs": 9007199254740993, + "method": "tools/call", + "protocol": "mcp", + "success": true, + "timestamp": "2026-08-18T14:33:41.882Z", + "toolName": "create_pull_request", + "type": "tool_call" + }, + "agent": { + "principal": "svc:agent-workspace", + "sessionId": "sess-4f2a" + }, + "chain": { + "previousHash": "genesis" + }, + "metadata": { + "attestorVersion": "example-gateway/0.4.0" + }, + "parties": [ + { + "party": "gateway", + "role": "witness", + "scope": [ + "toolName", + "timestamp", + "durationMs" + ] + } + ] + }, + "predicateType": "https://in-toto.io/attestation/ai-agent-action/v0.1", + "subject": [ + { + "digest": { + "sha256": "4e70d92204aee18162dc4db4d6bdc2462b73ec0c45d794ec5912bfec124cea3e" + }, + "name": "session:agent-workspace-4f2a" + } + ] +} diff --git a/atomic-canonical/tests/vectors/ingest/statements/vd94ac70c9f0d84bf.json b/atomic-canonical/tests/vectors/ingest/statements/vd94ac70c9f0d84bf.json new file mode 100644 index 00000000..f7eebf9f --- /dev/null +++ b/atomic-canonical/tests/vectors/ingest/statements/vd94ac70c9f0d84bf.json @@ -0,0 +1,303 @@ +{ + "_type": "https://in-toto.io/Statement/v1", + "predicate": { + "action": { + "durationMs": 412, + "method": "tools/call", + "protocol": "mcp", + "success": true, + "timestamp": "2026-08-18T14:33:41.882Z", + "toolName": "create_pull_request", + "type": "tool_call" + }, + "agent": { + "principal": "svc:agent-workspace", + "sessionId": "sess-4f2a" + }, + "chain": { + "previousHash": "genesis" + }, + "extensions": { + "n": { + "n": { + "n": { + "n": { + "n": { + "n": { + "n": { + "n": { + "n": { + "n": { + "n": { + "n": { + "n": { + "n": { + "n": { + "n": { + "n": { + "n": { + "n": { + "n": { + "n": { + "n": { + "n": { + "n": { + "n": { + "n": { + "n": { + "n": { + "n": { + "n": { + "n": { + "n": { + "n": { + "n": { + "n": { + "n": { + "n": { + "n": { + "n": { + "n": { + "n": { + "n": { + "n": { + "n": { + "n": { + "n": { + "n": { + "n": { + "n": { + "n": { + "n": { + "n": { + "n": { + "n": { + "n": { + "n": { + "n": { + "n": { + "n": { + "n": { + "n": { + "n": { + "n": { + "n": { + "n": { + "n": { + "n": { + "n": { + "n": { + "n": { + "n": { + "n": { + "n": { + "n": { + "n": { + "n": { + "n": { + "n": { + "n": { + "n": { + "n": { + "n": { + "n": { + "n": { + "n": { + "n": { + "n": { + "n": { + "n": { + "n": { + "n": { + "n": { + "n": { + "n": { + "n": { + "n": { + "n": { + "n": { + "n": { + "n": { + "n": { + "n": { + "n": { + "n": { + "n": { + "n": { + "n": { + "n": { + "n": { + "n": { + "n": { + "n": { + "n": { + "n": { + "n": { + "n": { + "n": { + "n": { + "n": { + "n": { + "n": { + "n": { + "n": { + "n": { + "n": { + "n": { + "n": { + "n": { + "n": "leaf" + } + } + } + } + } + } + } + } + } + } + } + } + } + } + } + } + } + } + } + } + } + } + } + } + } + } + } + } + } + } + } + } + } + } + } + } + } + } + } + } + } + } + } + } + } + } + } + } + } + } + } + } + } + } + } + } + } + } + } + } + } + } + } + } + } + } + } + } + } + } + } + } + } + } + } + } + } + } + } + } + } + } + } + } + } + } + } + } + } + } + } + } + } + } + } + } + } + } + } + } + } + } + } + } + } + } + } + } + } + } + } + } + } + } + } + } + } + } + } + } + } + } + } + } + } + } + } + } + }, + "metadata": { + "attestorVersion": "example-gateway/0.4.0" + }, + "parties": [ + { + "party": "gateway", + "role": "witness", + "scope": [ + "toolName", + "timestamp", + "durationMs" + ] + } + ] + }, + "predicateType": "https://in-toto.io/attestation/ai-agent-action/v0.1", + "subject": [ + { + "digest": { + "sha256": "4e70d92204aee18162dc4db4d6bdc2462b73ec0c45d794ec5912bfec124cea3e" + }, + "name": "session:agent-workspace-4f2a" + } + ] +} From 557cae2dfc64ae0c14b6efac84ff55bc28d00cc3 Mon Sep 17 00:00:00 2001 From: Sankalp Gilda Date: Sun, 20 Sep 2026 01:31:15 -0400 Subject: [PATCH 02/22] feat(cli): admit a grant at the boundary where its bytes arrive atomic identity grant reads a certificate from a file, from stdin or out of the identity store and then verifies it, so the bytes it parses are verification input and belong behind the same admission as the header and store paths in atomic-canonical. Route load, list, push, verify and revoke through jcs::admit_document, and delegate_from_request with them: grant new --request countersigns a self-signed request that arrived from the far end, and its signature is checked over the canonical form of whatever those bytes denote. No new dependency: the crate already depends on atomic-canonical, and jcs-admit stays a dependency of atomic-canonical alone. A document that fails admission is refused where it was previously parsed, which for load, verify, revoke and the request read is an error naming the fault and for list and push is the skip those paths already had. The typed reads elsewhere in the workspace are left alone on purpose. A struct deserialization already answers 'duplicate field' for a repeated member, while the same bytes into a serde_json::Value answer Ok and keep the last one, so the untyped reads are the ones that needed a decision. --- atomic-cli/src/commands/identity/delegate.rs | 11 ++++--- .../src/commands/identity/delegation.rs | 31 ++++++++++--------- 2 files changed, 23 insertions(+), 19 deletions(-) diff --git a/atomic-cli/src/commands/identity/delegate.rs b/atomic-cli/src/commands/identity/delegate.rs index 8fe0af08..a9e3f4fb 100644 --- a/atomic-cli/src/commands/identity/delegate.rs +++ b/atomic-cli/src/commands/identity/delegate.rs @@ -25,6 +25,7 @@ use std::path::PathBuf; use clap::Parser; use atomic_canonical::delegation as cert; +use atomic_canonical::jcs; use atomic_identity::delegation::{Delegation, DelegationScope}; use atomic_identity::{Identity, IdentityStore, IdentityType}; @@ -322,10 +323,12 @@ impl Delegate { std::fs::read_to_string(path)? }; - let value: serde_json::Value = - serde_json::from_str(&raw).map_err(|e| CliError::InvalidArgument { - message: format!("{path} is not valid JSON: {e}"), - })?; + // Admission before the value exists: a request arrives from the far end + // as bytes, and its self-signature is about to be checked over the + // canonical form of whatever those bytes denote. + let value = jcs::admit_document(raw.as_bytes()).map_err(|e| CliError::InvalidArgument { + message: format!("{path} was refused: {e}"), + })?; let request = cert::verify_request(&value).map_err(|e| CliError::DelegationError { message: format!( diff --git a/atomic-cli/src/commands/identity/delegation.rs b/atomic-cli/src/commands/identity/delegation.rs index 121e24b2..6df63cb3 100644 --- a/atomic-cli/src/commands/identity/delegation.rs +++ b/atomic-cli/src/commands/identity/delegation.rs @@ -26,6 +26,7 @@ use std::path::PathBuf; use clap::{Parser, Subcommand}; use atomic_canonical::delegation as cert; +use atomic_canonical::jcs; use atomic_identity::delegation::DelegationId; use atomic_identity::IdentityStore; use atomic_remote::storage_types::{PushDelegationRequest, RevokeDelegationRequest}; @@ -95,10 +96,12 @@ pub struct Install { impl Command for Install { fn run(&self) -> CliResult<()> { let raw = read_input(&self.path)?; - let value: serde_json::Value = - serde_json::from_str(&raw).map_err(|e| CliError::InvalidArgument { - message: format!("{} is not valid JSON: {e}", self.path), - })?; + // Admission before the value exists: the raw bytes are what gets stored + // and later re-verified, so a fault the parse would have swallowed is a + // fault this machine would keep. + let value = jcs::admit_document(raw.as_bytes()).map_err(|e| CliError::InvalidArgument { + message: format!("{} was refused: {e}", self.path), + })?; // Verify before storing. A certificate that does not verify is not // something to keep "in case" — it would be silently skipped at use @@ -187,8 +190,8 @@ impl Push { let mut pushed = 0; for (id, raw) in documents { - let Ok(value) = serde_json::from_str::(&raw) else { - print_warning(&format!("Skipping {id}: not valid JSON")); + let Ok(value) = jcs::admit_document(raw.as_bytes()) else { + print_warning(&format!("Skipping {id}: refused at the ingest boundary")); continue; }; let Ok(delegation) = cert::verify_self_contained(&value) else { @@ -253,7 +256,7 @@ impl Command for List { let mut any = false; for (id, raw) in stored { - let Ok(value) = serde_json::from_str::(&raw) else { + let Ok(value) = jcs::admit_document(raw.as_bytes()) else { continue; }; // Show what parses even if it does not verify — a certificate that @@ -337,10 +340,9 @@ impl Verify { load_document(&store, &self.target)? }; - let value: serde_json::Value = - serde_json::from_str(&raw).map_err(|e| CliError::InvalidArgument { - message: format!("not valid JSON: {e}"), - })?; + let value = jcs::admit_document(raw.as_bytes()).map_err(|e| CliError::InvalidArgument { + message: format!("refused: {e}"), + })?; let delegation = match cert::verify_self_contained(&value) { Ok(d) => { @@ -492,10 +494,9 @@ impl Revoke { .expect("clap requires one of id/--all-mine"); let id = normalize_id(raw_id)?; let raw = load_document(&store, raw_id)?; - let value: serde_json::Value = - serde_json::from_str(&raw).map_err(|e| CliError::InvalidArgument { - message: format!("stored grant is not valid JSON: {e}"), - })?; + let value = jcs::admit_document(raw.as_bytes()).map_err(|e| CliError::InvalidArgument { + message: format!("stored grant was refused: {e}"), + })?; let delegation = cert::parse(&value).map_err(|e| CliError::DelegationError { message: format!("stored grant is malformed: {e}"), })?; From 3e707a7dedcd1d9b3b7707adf3814012dc667999 Mon Sep 17 00:00:00 2001 From: Lee Faus Date: Mon, 21 Sep 2026 13:19:09 -0400 Subject: [PATCH 03/22] complete fix for file inode duplication (#206) --- atomic-core/src/pristine/traits/mutate.rs | 4 + atomic-core/src/pristine/txn/write/mod.rs | 22 + atomic-core/src/pristine/txn/write/tests.rs | 19 + .../src/repository/deferred_tree.rs | 517 ++++++++++-------- atomic-repository/src/repository/insert.rs | 98 ++-- atomic-repository/src/repository/mod.rs | 3 +- atomic-repository/src/repository/switch.rs | 17 +- .../tests/cross_view_merge_tests.rs | 94 +--- tests/harness/28_merge_rubric.sh | 13 +- tests/harness/39_ambient_inode_views.sh | 276 ++++++++++ tests/harness/40_switch_transactionality.sh | 81 +++ 11 files changed, 770 insertions(+), 374 deletions(-) create mode 100755 tests/harness/39_ambient_inode_views.sh create mode 100755 tests/harness/40_switch_transactionality.sh diff --git a/atomic-core/src/pristine/traits/mutate.rs b/atomic-core/src/pristine/traits/mutate.rs index 4e16c1a3..d6e2832d 100644 --- a/atomic-core/src/pristine/traits/mutate.rs +++ b/atomic-core/src/pristine/traits/mutate.rs @@ -202,6 +202,10 @@ pub trait MutTxnT: ViewTxnT + TreeTxnT + super::CrdtTxnT { /// Remove a file from the tree (removes path↔inode mappings). fn del_tree(&mut self, path: &str) -> Result, PristineError>; + /// Remove one inode's path binding without deleting a different inode that + /// currently occupies the same single-valued forward path entry. + fn del_tree_binding(&mut self, path: &str, inode: Inode) -> Result<(), PristineError>; + /// Store file index entry (mtime + size + content hash) for fast status detection. /// /// Called after a file is recorded or applied. Subsequent `status()` calls diff --git a/atomic-core/src/pristine/txn/write/mod.rs b/atomic-core/src/pristine/txn/write/mod.rs index 69428c71..0c52948a 100644 --- a/atomic-core/src/pristine/txn/write/mod.rs +++ b/atomic-core/src/pristine/txn/write/mod.rs @@ -1516,6 +1516,28 @@ impl<'a> MutTxnT for WriteTxn<'a> { Ok(inode) } + fn del_tree_binding(&mut self, path: &str, inode: Inode) -> PristineResult<()> { + { + let mut table = self.txn.open_table(TREE)?; + let owns_forward = table + .get(path)? + .is_some_and(|value| value.value() == inode.get()); + if owns_forward { + table.remove(path)?; + } + } + { + let mut table = self.txn.open_table(REV_TREE)?; + let matches_path = table + .get(inode.get())? + .is_some_and(|value| value.value() == path); + if matches_path { + table.remove(inode.get())?; + } + } + Ok(()) + } + fn put_file_index( &mut self, path: &str, diff --git a/atomic-core/src/pristine/txn/write/tests.rs b/atomic-core/src/pristine/txn/write/tests.rs index e56d35a4..46d26ae4 100644 --- a/atomic-core/src/pristine/txn/write/tests.rs +++ b/atomic-core/src/pristine/txn/write/tests.rs @@ -286,6 +286,25 @@ mod tests { txn.commit().unwrap(); } + #[test] + fn test_del_tree_binding_preserves_other_same_path_occupant() { + let dir = tempdir().unwrap(); + let db_path = dir.path().join("pristine"); + let pristine = Pristine::open(&db_path).unwrap(); + let mut txn = pristine.write_txn().unwrap(); + + let first = txn.alloc_inode().unwrap(); + let second = txn.alloc_inode().unwrap(); + txn.put_tree("same.txt", first).unwrap(); + txn.put_tree("same.txt", second).unwrap(); + + txn.del_tree_binding("same.txt", first).unwrap(); + + assert_eq!(txn.get_inode("same.txt").unwrap(), Some(second)); + assert_eq!(txn.get_path(first).unwrap(), None); + assert_eq!(txn.get_path(second).unwrap().as_deref(), Some("same.txt")); + } + #[test] fn test_graph_operations() { let dir = tempdir().unwrap(); diff --git a/atomic-repository/src/repository/deferred_tree.rs b/atomic-repository/src/repository/deferred_tree.rs index 9bc9028b..2669bf73 100644 --- a/atomic-repository/src/repository/deferred_tree.rs +++ b/atomic-repository/src/repository/deferred_tree.rs @@ -1,7 +1,7 @@ use super::*; use serde::{Deserialize, Serialize}; -use std::collections::{HashMap, HashSet}; +use std::collections::{BTreeSet, HashMap, HashSet}; use std::io::Write; const DEFERRED_TREE_JOURNAL: &str = "deferred-tree-ops.json"; @@ -31,10 +31,13 @@ pub(super) struct DeferredTreeOp { } #[derive(Debug, Clone, Serialize, Deserialize)] -struct DeferredTreeJournal { +pub(super) struct DeferredTreeJournal { version: u32, #[serde(default)] ops: Vec, + /// Projection-level causal predecessors captured at first ingestion. + #[serde(default)] + predecessors: HashMap>, } #[derive(Debug, Clone, Serialize, Deserialize)] @@ -49,85 +52,114 @@ impl Default for DeferredTreeJournal { Self { version: DEFERRED_TREE_JOURNAL_VERSION, ops: Vec::new(), + predecessors: HashMap::new(), } } } #[derive(Debug, Clone)] -struct DesiredTreePath { - desired_path: Option, - /// Journal order of the visible Set that currently claims this path. - /// A visible Set outranks an inherited baseline, and later visible Sets - /// resolve two view overlays that independently created the same path. - last_set_order: Option, +struct DesiredTreePaths { + paths: BTreeSet, } -type DeferredPathClaim = (Position, Option); - -fn desired_tree_paths( +fn desired_tree_paths( ops: &[DeferredTreeOp], + predecessors: &HashMap>, visible_changes: &HashSet, -) -> HashMap, DesiredTreePath> { - let mut desired = HashMap::new(); + mut depends_on: F, +) -> HashMap, DesiredTreePaths> +where + F: FnMut(Hash, Hash) -> bool, +{ + let mut baselines: HashMap, Option> = HashMap::new(); + let mut events: HashMap, Vec<(usize, &DeferredTreeOp)>> = HashMap::new(); for (order, op) in ops.iter().enumerate() { - let state = desired.entry(op.inode).or_insert_with(|| DesiredTreePath { - desired_path: op.baseline_path.clone(), - last_set_order: None, - }); - if !visible_changes.contains(&op.change) { - continue; + baselines + .entry(op.inode) + .or_insert_with(|| op.baseline_path.clone()); + if visible_changes.contains(&op.change) { + events.entry(op.inode).or_default().push((order, op)); } + } - match &op.action { - DeferredTreeAction::Set { path } => { - state.desired_path = Some(path.clone()); - state.last_set_order = Some(order); - } - DeferredTreeAction::Delete => { - state.desired_path = None; - state.last_set_order = None; - } + let mut desired = HashMap::new(); + for (inode, baseline) in baselines { + let visible = events.remove(&inode).unwrap_or_default(); + if visible.is_empty() { + desired.insert( + inode, + DesiredTreePaths { + paths: baseline.into_iter().collect(), + }, + ); + continue; } + + let maximal = visible.iter().filter(|(order, op)| { + !visible.iter().any(|(other_order, other)| { + (other.change == op.change && other_order > order) + || (other.change != op.change + && (depends_on(other.change, op.change) + || journal_depends_on(predecessors, other.change, op.change))) + }) + }); + let paths = maximal + .filter_map(|(_, op)| match &op.action { + DeferredTreeAction::Set { path } => Some(path.clone()), + DeferredTreeAction::Delete => None, + }) + .collect(); + desired.insert(inode, DesiredTreePaths { paths }); } - // TREE is a one-to-one path↔inode index, while two overlay-visible changes - // can independently add the same path. Match the lifecycle order Atomic - // established when those changes were recorded/imported: the latest - // visible Set owns the path. Baseline-only duplicates remain untouched so - // apply_deferred_tree_ops_in_txn still fails closed on corrupt TREE state. - let mut claims: HashMap> = HashMap::new(); - for (inode, state) in &desired { - if let Some(path) = &state.desired_path { - claims - .entry(path.clone()) - .or_default() - .push((*inode, state.last_set_order)); + desired +} + +fn journal_depends_on( + predecessors: &HashMap>, + change: Hash, + ancestor: Hash, +) -> bool { + let mut pending = vec![change]; + let mut visited = HashSet::new(); + while let Some(hash) = pending.pop() { + if !visited.insert(hash) { + continue; } - } - for claimants in claims.into_values().filter(|claimants| claimants.len() > 1) { - let Some(max_order) = claimants.iter().filter_map(|(_, order)| *order).max() else { + let Some(direct) = predecessors.get(&hash) else { continue; }; - if claimants - .iter() - .filter(|(_, order)| *order == Some(max_order)) - .count() - != 1 - { + if direct.contains(&ancestor) { + return true; + } + pending.extend(direct.iter().copied()); + } + false +} + +fn change_depends_on(txn: &T, change: Hash, ancestor: Hash) -> bool { + if change == ancestor { + return false; + } + let mut pending = vec![change]; + let mut visited = HashSet::new(); + while let Some(hash) = pending.pop() { + if !visited.insert(hash) { continue; } - for (inode, order) in claimants { - if order != Some(max_order) { - desired - .get_mut(&inode) - .expect("claimant exists") - .desired_path = None; - } + let Ok(Some(id)) = txn.get_internal(&hash) else { + continue; + }; + let Ok(deps) = txn.get_change_deps(id) else { + continue; + }; + if deps.contains(&ancestor) { + return true; } + pending.extend(deps); } - - desired + false } fn external_inode_position( @@ -160,25 +192,6 @@ fn push_unique(ops: &mut Vec, op: DeferredTreeOp) { } } -fn remember_op_paths(op: &DeferredTreeOp, paths: &mut HashSet) { - if let Some(path) = &op.baseline_path { - paths.insert(path.clone()); - } - if let DeferredTreeAction::Set { path } = &op.action { - paths.insert(path.clone()); - } -} - -fn op_touches_paths(op: &DeferredTreeOp, paths: &HashSet) -> bool { - op.baseline_path - .as_ref() - .is_some_and(|path| paths.contains(path)) - || matches!( - &op.action, - DeferredTreeAction::Set { path } if paths.contains(path) - ) -} - fn external_position( change_hash: Hash, position: Position>, @@ -211,37 +224,6 @@ fn current_path_for_position( .map_err(|e| RepositoryError::Database(e.to_string())) } -fn inode_is_visible_on_another_view( - txn: &T, - position: Position, - current_view: &str, -) -> Result { - let Some(internal_change) = txn - .get_internal(&position.change) - .map_err(|e| RepositoryError::Database(e.to_string()))? - else { - return Ok(false); - }; - for name in txn - .list_views() - .map_err(|e| RepositoryError::Database(e.to_string()))? - { - if name == current_view { - continue; - } - let Some(view) = txn - .get_view(&name) - .map_err(|e| RepositoryError::Database(e.to_string()))? - else { - continue; - }; - if collect_visible_change_ids(txn, &view)?.contains(&internal_change) { - return Ok(true); - } - } - Ok(false) -} - fn push_delete_for_path( txn: &T, change: Hash, @@ -272,7 +254,6 @@ fn push_occupant_baseline( activating_change: Hash, path: &str, exclude: Option>, - visible_on_target: Option<&HashSet>, ops: &mut Vec, ) -> Result<(), RepositoryError> { let Some(inode) = txn @@ -287,11 +268,7 @@ fn push_occupant_baseline( if Some(position) == exclude { return Ok(()); } - if visible_on_target.is_some_and(|visible| visible.contains(&position.change)) { - // The current occupant belongs to the target view; it is a real owner, - // not a foreign draft binding. Never manufacture a reverse delete. - return Ok(()); - } + push_unique( ops, DeferredTreeOp { @@ -313,7 +290,6 @@ pub(super) fn collect_tree_ops( change_hash: Hash, change: &Change, deleted_paths: &[String], - visible_on_target: Option<&HashSet>, ) -> Result, RepositoryError> { let mut ops = Vec::new(); @@ -330,14 +306,7 @@ pub(super) fn collect_tree_ops( // and switching back can restore it without overwriting // TREE/REV_TREE. let added_position = Position::new(change_hash, add_inode.start); - push_occupant_baseline( - txn, - change_hash, - path, - Some(added_position), - visible_on_target, - &mut ops, - )?; + push_occupant_baseline(txn, change_hash, path, Some(added_position), &mut ops)?; push_unique( &mut ops, DeferredTreeOp { @@ -352,14 +321,7 @@ pub(super) fn collect_tree_ops( let Some(external_position) = external_position(change_hash, add.inode) else { continue; }; - push_occupant_baseline( - txn, - change_hash, - path, - Some(external_position), - visible_on_target, - &mut ops, - )?; + push_occupant_baseline(txn, change_hash, path, Some(external_position), &mut ops)?; let op = DeferredTreeOp { change: change_hash, inode: external_position, @@ -429,7 +391,9 @@ impl Repository { self.dot_dir.join(DEFERRED_TREE_JOURNAL) } - fn load_deferred_tree_journal(&self) -> Result { + pub(super) fn load_deferred_tree_journal( + &self, + ) -> Result { let path = self.deferred_tree_journal_path(); if !path.is_file() { return Ok(DeferredTreeJournal::default()); @@ -446,12 +410,11 @@ impl Repository { /// Persist deferred operations atomically. The caller holds pristine's /// write transaction, which serializes journal writers across processes. - pub(super) fn append_deferred_tree_ops( + pub(super) fn append_deferred_tree_ops( &self, txn: &T, ops: &[DeferredTreeOp], current_view: &str, - include_new_inodes: bool, ) -> Result<(), RepositoryError> { if ops.is_empty() { return Ok(()); @@ -459,42 +422,47 @@ impl Repository { let mut journal = self.load_deferred_tree_journal()?; let mut changed = false; - let mut shared_creators = HashMap::new(); - let tracked_inodes: HashSet> = - journal.ops.iter().map(|op| op.inode).collect(); - let mut tracked_paths = HashSet::new(); - for op in &journal.ops { - remember_op_paths(op, &mut tracked_paths); - } - let mut batch_participates = include_new_inodes; - if !batch_participates { - for op in ops { - let inode_is_tracked = tracked_inodes.contains(&op.inode); - let path_is_tracked = op_touches_paths(op, &tracked_paths); - let inode_is_shared = if inode_is_tracked || path_is_tracked { - false - } else if let Some(shared) = shared_creators.get(&op.inode.change) { - *shared - } else { - let shared = inode_is_visible_on_another_view(txn, op.inode, current_view)?; - shared_creators.insert(op.inode.change, shared); - shared - }; - if inode_is_tracked || path_is_tracked || inode_is_shared { - batch_participates = true; - break; - } + let view = txn + .get_view(current_view) + .map_err(|e| RepositoryError::Database(e.to_string()))? + .ok_or_else(|| RepositoryError::ViewNotFound { + name: current_view.to_string(), + })?; + let visible_ids = collect_visible_change_ids(txn, &view)?; + let mut visible_hashes = HashSet::with_capacity(visible_ids.len()); + for id in visible_ids { + if let Some(hash) = txn + .get_external(id) + .map_err(|e| RepositoryError::Database(e.to_string()))? + { + visible_hashes.insert(hash); } } - if !batch_participates { - return Ok(()); - } // A Change is Atomic's visibility unit. Once one path operation joins // a deferred lifecycle, retain every path operation from that change; // otherwise a rename encoded as tracked-delete + new-inode-add would // lose its destination half during replay. for op in ops { + if matches!(op.action, DeferredTreeAction::Set { .. }) { + let mut predecessors: Vec = journal + .ops + .iter() + .filter(|existing| { + existing.inode == op.inode + && existing.change != op.change + && visible_hashes.contains(&existing.change) + && matches!(existing.action, DeferredTreeAction::Set { .. }) + }) + .map(|existing| existing.change) + .collect(); + predecessors.sort(); + predecessors.dedup(); + if journal.predecessors.get(&op.change) != Some(&predecessors) { + journal.predecessors.insert(op.change, predecessors); + changed = true; + } + } if let Some(existing) = journal.ops.iter_mut().find(|existing| { existing.change == op.change && existing.inode == op.inode @@ -585,7 +553,7 @@ impl Repository { Ok(lock) } - fn clear_deferred_tree_alignment_pending(&self) -> Result<(), RepositoryError> { + pub(super) fn clear_deferred_tree_alignment_pending(&self) -> Result<(), RepositoryError> { match std::fs::remove_file(self.deferred_tree_alignment_pending_path()) { Ok(()) => self.sync_dot_dir(), Err(error) if error.kind() == std::io::ErrorKind::NotFound => Ok(()), @@ -597,7 +565,7 @@ impl Repository { self.deferred_tree_alignment_pending_path().is_file() } - fn apply_deferred_tree_ops_in_txn( + pub(super) fn apply_deferred_tree_ops_in_txn( &self, txn: &mut atomic_core::pristine::WriteTxn<'_>, journal: &DeferredTreeJournal, @@ -620,8 +588,21 @@ impl Repository { } } - let desired = desired_tree_paths(&journal.ops, &visible_hashes); - let mut updates = Vec::new(); + let desired = desired_tree_paths( + &journal.ops, + &journal.predecessors, + &visible_hashes, + |change, ancestor| change_depends_on(&*txn, change, ancestor), + ); + let mut current_paths: HashMap> = HashMap::new(); + for entry in txn + .iter_tree() + .map_err(|e| RepositoryError::Database(e.to_string()))? + { + let (path, inode) = entry.map_err(|e| RepositoryError::Database(e.to_string()))?; + current_paths.entry(inode).or_default().push(path); + } + let mut bindings = Vec::new(); let mut affected_paths = HashSet::new(); for (external_position, state) in desired { let Some(internal_change) = txn @@ -636,52 +617,28 @@ impl Repository { else { continue; }; - let current_path = txn - .get_path(inode) - .map_err(|e| RepositoryError::Database(e.to_string()))?; - if current_path == state.desired_path { - continue; - } - if let Some(path) = ¤t_path { - affected_paths.insert(path.clone()); - } - if let Some(path) = &state.desired_path { - affected_paths.insert(path.clone()); - } - updates.push((inode, current_path, state.desired_path)); + let current = current_paths.remove(&inode).unwrap_or_default(); + affected_paths.extend(current.iter().cloned()); + affected_paths.extend(state.paths.iter().cloned()); + bindings.push((external_position, inode, current, state.paths)); } - // Remove all stale sources first so rename chains and swaps are safe. - for (_, current_path, _) in &updates { - if let Some(path) = current_path { - txn.del_tree(path) + // Remove each inode's own reverse claim and only remove the forward + // entry when that inode is its current occupant. Deleting by path alone + // can unbind a different same-name inode. + for (_, inode, current_paths, _) in &bindings { + for path in current_paths { + txn.del_tree_binding(path, *inode) .map_err(|e| RepositoryError::Database(e.to_string()))?; } } - // Never overwrite an unrelated destination: put_tree would update - // TREE but leave the old occupant's REV_TREE entry stale. - for (inode, _, desired_path) in &updates { - let Some(path) = desired_path else { - continue; - }; - if let Some(occupant) = txn - .get_inode(path) - .map_err(|e| RepositoryError::Database(e.to_string()))? - { - if occupant != *inode { - return Err(RepositoryError::InvalidOperation { - message: format!( - "cannot replay deferred path to '{}': path is owned by another inode", - path - ), - }); - } - } - } - - for (inode, _, desired_path) in updates { - if let Some(path) = desired_path { + // A path may have multiple visible inode claims. Reinsert all of them + // in stable identity order: TREE retains one deterministic lookup + // occupant while REV_TREE preserves every claim for conflict detection. + bindings.sort_by_key(|(position, _, _, _)| *position); + for (_, inode, _, desired_paths) in bindings { + for path in desired_paths { txn.put_tree(&path, inode) .map_err(|e| RepositoryError::Database(e.to_string()))?; } @@ -689,6 +646,21 @@ impl Repository { Ok(affected_paths) } + pub(super) fn refresh_deferred_tree_projection( + &self, + view_name: &str, + ) -> Result, RepositoryError> { + let mut txn = self + .pristine + .write_txn() + .map_err(|e| RepositoryError::Database(e.to_string()))?; + let journal = self.load_deferred_tree_journal()?; + let affected = self.apply_deferred_tree_ops_in_txn(&mut txn, &journal, view_name)?; + txn.commit() + .map_err(|e| RepositoryError::Database(e.to_string()))?; + Ok(affected) + } + /// Recover a switch interrupted between TREE alignment and publishing the /// current-view pointer. The advisory lock is released by the OS on process /// exit, so a concurrent opener waits for a live switch and only performs @@ -705,6 +677,11 @@ impl Repository { return Ok(()); } let pending = self.load_deferred_tree_alignment_pending()?; + // Capture both view path sets before restoring the source projection; + // TREE is a selected-view index and target-only paths may no longer be + // discoverable after alignment. + let source_files = self.visible_file_paths(&pending.source_view)?; + let target_files = self.visible_file_paths(&pending.target_view)?; let mut txn = self .pristine .write_txn() @@ -714,7 +691,23 @@ impl Repository { self.write_current_view(&pending.source_view)?; txn.commit() .map_err(|e| RepositoryError::Database(e.to_string()))?; - self.current_view = pending.source_view; + self.current_view = pending.source_view.clone(); + + // The marker spans filesystem materialization as well as TREE/pointer + // publication. Remove any target-only tracked files that may have been + // written before interruption, then reconstruct the source view. If + // either step fails, retain the marker for the next writable open. + for path in target_files.difference(&source_files) { + let abs = self.root.join(path); + if abs.is_file() { + std::fs::remove_file(&abs)?; + } + } + self.materialize()?; + let source_workspace = workspace_path(&self.dot_dir, &pending.source_view); + if source_workspace.is_dir() { + self.restore_workspace_to_working_copy(&source_workspace); + } self.clear_deferred_tree_alignment_pending()?; Ok(()) } @@ -726,6 +719,7 @@ impl Repository { pub(super) fn align_deferred_tree_and_publish_view( &mut self, view_name: &str, + retain_marker_for_materialization: bool, ) -> Result, RepositoryError> { let _alignment_lock = self.lock_deferred_tree_alignment()?; // `current_view` can intentionally be scoped to a background target @@ -767,10 +761,9 @@ impl Repository { } self.current_view = view_name.to_string(); - // Clearing the marker is the commit point for this recoverable - // transition. Propagate cleanup or directory-sync failures so a switch - // is never reported durable while recovery may still roll it back. - self.clear_deferred_tree_alignment_pending()?; + if !retain_marker_for_materialization { + self.clear_deferred_tree_alignment_pending()?; + } Ok(affected_paths) } } @@ -808,12 +801,14 @@ mod tests { }, ]; - let base = desired_tree_paths(&ops, &HashSet::new()); - assert_eq!(base[&inode].desired_path.as_deref(), Some("a.txt")); + let base = desired_tree_paths(&ops, &HashMap::new(), &HashSet::new(), |_, _| false); + assert_eq!(base[&inode].paths, BTreeSet::from(["a.txt".into()])); let visible = HashSet::from([first, second]); - let moved = desired_tree_paths(&ops, &visible); - assert_eq!(moved[&inode].desired_path.as_deref(), Some("c.txt")); + let moved = desired_tree_paths(&ops, &HashMap::new(), &visible, |change, ancestor| { + change == second && ancestor == first + }); + assert_eq!(moved[&inode].paths, BTreeSet::from(["c.txt".into()])); } #[test] @@ -838,8 +833,13 @@ mod tests { }, ]; - let desired = desired_tree_paths(&ops, &HashSet::from([moved, deleted])); - assert_eq!(desired[&inode].desired_path, None); + let desired = desired_tree_paths( + &ops, + &HashMap::new(), + &HashSet::from([moved, deleted]), + |change, ancestor| change == deleted && ancestor == moved, + ); + assert!(desired[&inode].paths.is_empty()); } #[test] @@ -868,15 +868,59 @@ mod tests { }, ]; - let source = desired_tree_paths(&ops, &HashSet::from([foreground])); - assert_eq!(source[&inode].desired_path.as_deref(), Some("source.txt")); + let source = desired_tree_paths( + &ops, + &HashMap::new(), + &HashSet::from([foreground]), + |_, _| false, + ); + assert_eq!(source[&inode].paths, BTreeSet::from(["source.txt".into()])); + + let target = + desired_tree_paths(&ops, &HashMap::new(), &HashSet::from([deferred]), |_, _| { + false + }); + assert_eq!(target[&inode].paths, BTreeSet::from(["target.txt".into()])); + } + + #[test] + fn planner_preserves_concurrent_rename_destinations_for_same_inode() { + let inode = Position::new(hash("creator"), ChangePosition::new(13)); + let left = hash("rename-left"); + let right = hash("rename-right"); + let ops = vec![ + DeferredTreeOp { + change: left, + inode, + baseline_path: Some("original.txt".into()), + action: DeferredTreeAction::Set { + path: "left.txt".into(), + }, + }, + DeferredTreeOp { + change: right, + inode, + baseline_path: Some("original.txt".into()), + action: DeferredTreeAction::Set { + path: "right.txt".into(), + }, + }, + ]; - let target = desired_tree_paths(&ops, &HashSet::from([deferred])); - assert_eq!(target[&inode].desired_path.as_deref(), Some("target.txt")); + let desired = desired_tree_paths( + &ops, + &HashMap::new(), + &HashSet::from([left, right]), + |_, _| false, + ); + assert_eq!( + desired[&inode].paths, + BTreeSet::from(["left.txt".into(), "right.txt".into()]) + ); } #[test] - fn planner_chooses_latest_visible_inode_for_same_path() { + fn planner_preserves_all_visible_inodes_for_same_path() { let target_inode = Position::new(hash("target-creator"), ChangePosition::new(3)); let source_inode = Position::new(hash("source-creator"), ChangePosition::new(5)); let target_add = hash("target-add"); @@ -900,18 +944,31 @@ mod tests { }, ]; - let target = desired_tree_paths(&ops, &HashSet::from([target_add])); + let target = desired_tree_paths( + &ops, + &HashMap::new(), + &HashSet::from([target_add]), + |_, _| false, + ); assert_eq!( - target[&target_inode].desired_path.as_deref(), - Some("same.txt") + target[&target_inode].paths, + BTreeSet::from(["same.txt".into()]) ); - assert_eq!(target[&source_inode].desired_path, None); + assert!(target[&source_inode].paths.is_empty()); - let overlay = desired_tree_paths(&ops, &HashSet::from([target_add, source_add])); - assert_eq!(overlay[&target_inode].desired_path, None); + let overlay = desired_tree_paths( + &ops, + &HashMap::new(), + &HashSet::from([target_add, source_add]), + |_, _| false, + ); + assert_eq!( + overlay[&target_inode].paths, + BTreeSet::from(["same.txt".into()]) + ); assert_eq!( - overlay[&source_inode].desired_path.as_deref(), - Some("same.txt") + overlay[&source_inode].paths, + BTreeSet::from(["same.txt".into()]) ); } } diff --git a/atomic-repository/src/repository/insert.rs b/atomic-repository/src/repository/insert.rs index e4168eae..32ea85e5 100644 --- a/atomic-repository/src/repository/insert.rs +++ b/atomic-repository/src/repository/insert.rs @@ -786,7 +786,7 @@ impl Repository { txn.put_change_deps(change_id, final_change.dependencies()) .map_err(|e| RepositoryError::Database(e.to_string()))?; - let tree_ops = collect_tree_ops(&txn, hash, &final_change, deleted_paths, None)?; + let tree_ops = collect_tree_ops(&txn, hash, &final_change, deleted_paths)?; for graph_op in final_change.hunks() { match graph_op { @@ -903,7 +903,7 @@ impl Repository { timings.direct_crdt_ms = direct_crdt_ms; let commit_start = std::time::Instant::now(); - self.append_deferred_tree_ops(&txn, &tree_ops, view_name, preserve_existing_tree_paths)?; + self.append_deferred_tree_ops(&txn, &tree_ops, view_name)?; txn.commit() .map_err(|e| RepositoryError::Database(e.to_string()))?; timings.commit_ms = commit_start.elapsed().as_millis(); @@ -1011,7 +1011,7 @@ impl Repository { txn.put_change_deps(change_id, final_change.dependencies()) .map_err(|e| RepositoryError::Database(e.to_string()))?; - let tree_ops = collect_tree_ops(&txn, hash, &final_change, deleted_paths, None)?; + let tree_ops = collect_tree_ops(&txn, hash, &final_change, deleted_paths)?; let apply_start = std::time::Instant::now(); let insert = if import_graph_first_can_apply(&final_change) { @@ -1052,7 +1052,7 @@ impl Repository { } let commit_start = std::time::Instant::now(); - self.append_deferred_tree_ops(&txn, &tree_ops, view_name, preserve_existing_tree_paths)?; + self.append_deferred_tree_ops(&txn, &tree_ops, view_name)?; txn.commit() .map_err(|e| RepositoryError::Database(e.to_string()))?; timings.commit_ms = commit_start.elapsed().as_millis(); @@ -1064,6 +1064,7 @@ impl Repository { }) } + /// Save and apply an already-built git-import graph change. fn write_import_graph_first_direct( &self, txn: &mut atomic_core::pristine::WriteTxn<'_>, @@ -1734,28 +1735,15 @@ impl Repository { already_in_graph, change.hunks().len() ); - // External hashes of the target view's effective visible set. Occupant - // baselines must not manufacture a reverse delete of an inode whose - // introducing change the target view can already see: an - // already-ambient change being selected in is an earlier/concurrent - // event, never authorization to unbind a live owner. - let target_view = txn - .get_view(view_name) - .map_err(|e| RepositoryError::Database(e.to_string()))? - .ok_or_else(|| RepositoryError::ViewNotFound { - name: view_name.to_string(), - })?; - let visible = collect_visible_change_ids_with_deps(&txn, &target_view)?; - let mut visible_hashes: HashSet = HashSet::with_capacity(visible.len()); - for id in &visible { - if let Some(hash) = txn - .get_external(*id) - .map_err(|e| RepositoryError::Database(e.to_string()))? - { - visible_hashes.insert(hash); - } - } - let tree_ops = collect_tree_ops(&txn, *hash, &change, &[], Some(&visible_hashes))?; + // Lifecycle metadata is captured once when a change first enters the + // ambient graph. Selecting an existing change into another view only + // expands that view's closure; projection refresh consumes the + // canonical journal without deriving new operations from active TREE. + let tree_ops = if already_in_graph { + Vec::new() + } else { + collect_tree_ops(&txn, *hash, &change, &[])? + }; // Populate tree tables for FileAdd/DirAdd/FileDel hunks. // This creates the path→inode→position mappings that materialize @@ -1796,16 +1784,7 @@ impl Repository { txn.put_directory(new_inode, directory_flags::explicit_empty()) .map_err(|e| RepositoryError::Database(e.to_string()))?; } - GraphOp::FileDel { path, .. } if !preserve_existing_tree_paths => { - // View-aware: only remove TREE entry when no other - // view still references the file's creating change. - if let Ok(Some(inode)) = txn.get_inode(path) { - let dominated = is_file_only_on_view(&txn, inode, view_name); - if dominated { - let _ = txn.del_tree(path); - } - } - } + // NOTE: FileMove TREE maintenance is handled unconditionally // below (not gated by `!already_in_graph`), because a // draft-recorded rename inserted cross-view is always @@ -1848,15 +1827,10 @@ impl Repository { let inode_pos = Position::new(inode_change_id, add.inode.pos); if let Ok(Some(inode)) = txn.position_inode(inode_pos) { if let Ok(Some(old_path)) = txn.get_path(inode) { - // Only repoint when the tracked path actually differs - // (guards against a prior FileMove in this same change - // already having updated it). if old_path != *path { - let _ = txn.del_tree(&old_path); moved_from_disk.push(old_path); } } - let _ = txn.put_tree(path, inode); } } } @@ -1887,7 +1861,7 @@ impl Repository { // Commit the transaction log::debug!("insert_change: committing transaction..."); let commit_start = std::time::Instant::now(); - self.append_deferred_tree_ops(&txn, &tree_ops, view_name, preserve_existing_tree_paths)?; + self.append_deferred_tree_ops(&txn, &tree_ops, view_name)?; txn.commit() .map_err(|e| RepositoryError::Database(e.to_string()))?; let commit_ms = commit_start.elapsed().as_millis(); @@ -1908,6 +1882,28 @@ impl Repository { log::debug!("insert_change: txn.commit() took {}ms", commit_ms); } + // Refresh the active TREE projection for structural operations. A + // rename always changes path ownership. A delete does so only when + // graph retrieval confirms that concurrent edits did not keep the + // same inode alive on the target view. + if view_name == self.current_view { + let has_add = change + .hunks() + .iter() + .any(|op| matches!(op, GraphOp::FileAdd { .. } | GraphOp::DirAdd { .. })); + let has_move = change + .hunks() + .iter() + .any(|op| matches!(op, GraphOp::FileMove { .. })); + let has_effective_delete = change.hunks().iter().any(|op| { + matches!(op, GraphOp::FileDel { path, .. } + if matches!(self.get_file_content_on_view(path, view_name), Ok(None))) + }); + if has_add || has_move || has_effective_delete { + self.refresh_deferred_tree_projection(view_name)?; + } + } + // Working-copy cleanup for whole-file deletions (FileDel hunks). // // TREE/INODES are global and only cleaned up when no other view still @@ -1933,10 +1929,11 @@ impl Repository { } } - // Remove the stale source of each applied FileMove. TREE was - // repointed old→new above, so materialize will write the new path - // but never deletes the old one. Only remove when the old path is - // truly untracked on this view now (guards an A12-style shared path). + // Refresh each FileMove source. If another inode still claims the + // old path, rewrite it from that surviving identity; otherwise + // remove the stale file. Treating every move as path deletion would + // erase a sibling inode in a same-name conflict. + let mut paths_to_refresh = HashSet::new(); for old_path in &moved_from_disk { if matches!(self.get_file_inode(old_path), Ok(None)) { let abs = self.root.join(old_path); @@ -1944,8 +1941,13 @@ impl Repository { let _ = std::fs::remove_file(&abs); } let _ = self.del_file_index(old_path); + } else { + paths_to_refresh.insert(old_path.clone()); } } + if !paths_to_refresh.is_empty() { + self.materialize_paths(paths_to_refresh)?; + } } if trace_insert { @@ -2175,7 +2177,7 @@ impl Repository { // Determine which view to use let view_name = options.view.as_deref().unwrap_or(&self.current_view); let preserve_existing_tree_paths = view_name != self.current_view; - let tree_ops = collect_tree_ops(&txn, *hash, change, outcome.deleted_files(), None)?; + let tree_ops = collect_tree_ops(&txn, *hash, change, outcome.deleted_files())?; // Before applying atoms, set up tree entries for FileAdd hunks. // This creates the inode→position and path→inode mappings needed @@ -2314,7 +2316,7 @@ impl Repository { // Commit the transaction let commit_start = std::time::Instant::now(); - self.append_deferred_tree_ops(&txn, &tree_ops, view_name, preserve_existing_tree_paths)?; + self.append_deferred_tree_ops(&txn, &tree_ops, view_name)?; txn.commit() .map_err(|e| RepositoryError::Database(e.to_string()))?; if trace_record { diff --git a/atomic-repository/src/repository/mod.rs b/atomic-repository/src/repository/mod.rs index a92220b5..4c98e8fe 100644 --- a/atomic-repository/src/repository/mod.rs +++ b/atomic-repository/src/repository/mod.rs @@ -773,7 +773,8 @@ default = "{}" /// Returns an error if the view does not exist or the pointer file /// cannot be written. pub fn align_to_view(&mut self, view: &str) -> Result<(), RepositoryError> { - self.align_deferred_tree_and_publish_view(view).map(|_| ()) + self.align_deferred_tree_and_publish_view(view, false) + .map(|_| ()) } /// Set the current view on this handle only. diff --git a/atomic-repository/src/repository/switch.rs b/atomic-repository/src/repository/switch.rs index 837823a9..0b6bfd23 100644 --- a/atomic-repository/src/repository/switch.rs +++ b/atomic-repository/src/repository/switch.rs @@ -97,7 +97,7 @@ impl Repository { // Apply only the small set of view-scoped TREE operations and publish // the new pointer while holding the same database write lock. A marker // makes the transition recoverable if the process exits mid-switch. - let deferred_paths = self.align_deferred_tree_and_publish_view(view)?; + let deferred_paths = self.align_deferred_tree_and_publish_view(view, true)?; // Compute files visible on the NEW view. let new_files = self.visible_file_paths(view)?; @@ -223,6 +223,12 @@ impl Repository { .chain(ignored_paths.iter().map(|s| s.as_str())); cleanup_empty_ancestors(&self.root, all_removed); + if std::env::var_os("ATOMIC_TEST_FAIL_SWITCH_AFTER_PUBLISH").is_some() { + return Err(RepositoryError::InvalidOperation { + message: "injected switch failure after view publication".to_string(), + }); + } + // ── Phase 4: Materialize the new view's tracked files from graph ─ // // Instead of materializing ALL files, compute which files differ @@ -287,6 +293,12 @@ impl Repository { } }; + if std::env::var_os("ATOMIC_TEST_FAIL_SWITCH_AFTER_MATERIALIZE").is_some() { + return Err(RepositoryError::InvalidOperation { + message: "injected switch failure after materialization".to_string(), + }); + } + // ── Phase 5: Restore ignored files from the NEW view's workspace ─ // // Move artifacts from `.atomic/workspaces//` back into @@ -296,6 +308,7 @@ impl Repository { self.restore_workspace_to_working_copy(&new_ws); } + self.clear_deferred_tree_alignment_pending()?; Ok(result) } @@ -304,7 +317,7 @@ impl Repository { /// Walks the top-level entries in `ws_dir` and moves each into the /// project root via `rename()`. Skips the `.atomic` directory if /// present. - fn restore_workspace_to_working_copy(&self, ws_dir: &Path) { + pub(super) fn restore_workspace_to_working_copy(&self, ws_dir: &Path) { let entries = match std::fs::read_dir(ws_dir) { Ok(e) => e, Err(_) => return, diff --git a/atomic-repository/src/repository/tests/cross_view_merge_tests.rs b/atomic-repository/src/repository/tests/cross_view_merge_tests.rs index 6b0b8429..113bc4bd 100644 --- a/atomic-repository/src/repository/tests/cross_view_merge_tests.rs +++ b/atomic-repository/src/repository/tests/cross_view_merge_tests.rs @@ -892,35 +892,12 @@ fn test_cross_view_merge_post_merge_record_clean() { ); } -/// Regression (B2): selecting an earlier change into a view must not record a -/// reverse `Delete` of the view's current occupant when that occupant is -/// already visible on the target. -/// -/// Scenario: -/// - `earlier` (draft from dev): writes f.txt = "generation one", records. -/// - Switch to dev: the shared TREE keeps generation one's stale binding. -/// - `target` (a sibling of `earlier`, from dev): overwrites f.txt with -/// "generation two" and records — so generation two is the live occupant -/// of f.txt on `target`. -/// -/// Bug (pre-fix): collecting TREE ops for the earlier change (generation one) -/// while it is selected into `target` manufactured a reverse `Delete` of the -/// current occupant (generation two, visible on `target`). Replaying that -/// journal unbound f.txt, losing the live owner. -/// -/// Fixed: `push_occupant_baseline` skips the reverse delete when the -/// occupant's introducing change is visible on the target. -/// -/// This test checks the direct, replay-independent effect: the tree ops -/// collected for the earlier change must not contain a `Delete` of any -/// target-visible change. +/// Selecting an already-ambient change expands the target closure without +/// appending projection-derived lifecycle operations. #[test] -fn cross_view_insert_does_not_record_reverse_delete_of_visible_occupant() { - use super::deferred_tree::{collect_tree_ops, DeferredTreeAction}; - +fn cross_view_insert_does_not_append_deferred_tree_ops() { let (temp_dir, mut repo) = create_temp_repo(); - // Generation one on `earlier` (draft from dev). repo.create_view_from("earlier", "dev").unwrap(); repo.switch_view("earlier").unwrap(); let file = temp_dir.path().join("f.txt"); @@ -928,71 +905,18 @@ fn cross_view_insert_does_not_record_reverse_delete_of_visible_occupant() { repo.add("f.txt", TrackingOptions::default()).unwrap(); record_all(&repo, "generation one"); - // Switch to the sibling root; the shared TREE keeps gen1's stale binding. repo.switch_view("dev").unwrap(); - - // Generation two on `target` (a sibling of `earlier`, from dev). After this - // record, generation two is the live occupant of f.txt on `target`. repo.create_view_from("target", "dev").unwrap(); repo.switch_view("target").unwrap(); std::fs::write(&file, "generation two\n").unwrap(); repo.add("f.txt", TrackingOptions::default()).unwrap(); record_all(&repo, "generation two"); - // gen1 = the latest change on `earlier`; gen2 = the latest change on - // `target` (the current occupant of f.txt). - let gen1_hash = repo - .get_view_changes(Some("earlier")) - .unwrap() - .into_iter() - .max_by_key(|(seq, _)| *seq) - .unwrap() - .1; - let gen2_hash = repo - .get_view_changes(Some("target")) - .unwrap() - .into_iter() - .max_by_key(|(seq, _)| *seq) - .unwrap() - .1; - let gen1_change = repo.load_change(&gen1_hash).unwrap(); - - // The set of changes visible on target (includes gen2, the live occupant). - let target_visible: std::collections::HashSet = repo - .get_view_changes(Some("target")) - .unwrap() - .into_iter() - .map(|(_, h)| h) - .collect(); - assert!( - target_visible.contains(&gen2_hash), - "gen2 must be visible on target" - ); - - let txn = repo.pristine.read_txn().unwrap(); - - // The B2 fix: when the current occupant (gen2) is visible on target, - // collecting tree ops for the earlier change (gen1) must NOT record a - // reverse Delete of any target-visible change. - let ops_visible = - collect_tree_ops(&txn, gen1_hash, &gen1_change, &[], Some(&target_visible)).unwrap(); - assert!( - !ops_visible - .iter() - .any(|op| matches!(op.action, DeferredTreeAction::Delete) - && target_visible.contains(&op.inode.change)), - "B2: must not record a reverse Delete of a target-visible occupant" - ); + let journal = repo.dot_dir.join("deferred-tree-ops.json"); + let before = std::fs::read(&journal).unwrap(); + repo.insert_from_view(CrossViewInsertOptions::new("earlier", "target")) + .unwrap(); + let after = std::fs::read(&journal).unwrap(); - // Control: without the visibility set, the occupant-baseline Delete of gen2 - // IS recorded (the pre-fix behavior). - let empty: std::collections::HashSet = std::collections::HashSet::new(); - let ops_unchecked = collect_tree_ops(&txn, gen1_hash, &gen1_change, &[], Some(&empty)).unwrap(); - assert!( - ops_unchecked - .iter() - .any(|op| matches!(op.action, DeferredTreeAction::Delete) - && op.inode.change == gen2_hash), - "control: without visibility, the occupant Delete of gen2 is recorded" - ); + assert_eq!(after, before, "insert must consume canonical lifecycle metadata without appending target-specific operations"); } diff --git a/tests/harness/28_merge_rubric.sh b/tests/harness/28_merge_rubric.sh index cbc4007a..37d9b479 100755 --- a/tests/harness/28_merge_rubric.sh +++ b/tests/harness/28_merge_rubric.sh @@ -439,14 +439,11 @@ switch_view "$BASE_VIEW" >/dev/null mv orig.txt base-name.txt record_change "rename orig->base-name" >/dev/null atomic insert "$A11_HASH" >/dev/null 2>&1 -# Correct = a surfaced name conflict, OR both destination names preserved. -pred_a11_name_conflict_surfaced() { - if atomic conflicts --short 2>/dev/null | grep -qE ':'; then - return 0 # some conflict surfaced - fi - [[ -f feat-name.txt && -f base-name.txt ]] # both names preserved -} -xfail_correct "A11: rename-vs-rename surfaces a name conflict" pred_a11_name_conflict_surfaced +# Concurrent destinations are incomparable, so both names remain live and +# materialize the same stable inode content. +assert_file_content "A11: feature destination is preserved" feat-name.txt "shared content" +assert_file_content "A11: base destination is preserved" base-name.txt "shared content" +assert_file_not_exists "A11: superseded original path is absent" orig.txt if [[ "${KNOWN_BUGS:-0}" -gt 0 ]]; then echo "" diff --git a/tests/harness/39_ambient_inode_views.sh b/tests/harness/39_ambient_inode_views.sh new file mode 100755 index 00000000..ff929fd0 --- /dev/null +++ b/tests/harness/39_ambient_inode_views.sh @@ -0,0 +1,276 @@ +#!/usr/bin/env bash +# 39_ambient_inode_views.sh — Ambient inode identity across view closures. +# +# Covers the causal distinctions that TREE projection must not blur: +# 1. A child draft inherits a file introduced by its draft parent. +# 2. Selecting an already-ambient change is idempotent and does not append +# lifecycle operations derived from the active TREE projection. +# 3. Sibling drafts may independently create the same path; combining them +# surfaces both identities honestly instead of silently choosing an owner. + +HARNESS_DIR="$(cd "$(dirname "$0")" && pwd)" +source "$HARNESS_DIR/helpers.sh" +source "$HARNESS_DIR/merge_helpers.sh" + +journal_fingerprint() { + if [[ -f .atomic/deferred-tree-ops.json ]]; then + cksum .atomic/deferred-tree-ops.json + else + printf 'missing\n' + fi +} + +assert_equal_value() { + local desc="$1" + local expected="$2" + local actual="$3" + if [[ "$actual" == "$expected" ]]; then + _pass "$desc" + else + _fail "$desc" "expected '$expected', got '$actual'" + fi +} + + + +# ── Case 1: draft parent closure is inherited by its child ────────────────── +begin_section "Descendant draft inherits parent file identity" +make_temp_repo ambient-inode-descendant +init_repo + +new_view ab12 --draft --parent dev --switch >/dev/null +printf 'from-ab12\n' > f.txt +add_files f.txt >/dev/null +record_change "ab12 adds f.txt" >/dev/null + +new_view cd23 --draft --parent ab12 --switch >/dev/null +assert_file_content "cd23 materializes ab12's inherited file" f.txt "from-ab12" +assert_clean "cd23 starts clean with inherited file" + +ADD_OUTPUT="$(atomic add f.txt 2>&1)" +assert_clean "adding inherited file is idempotent" +assert_file_content "idempotent add preserves inherited content" f.txt "from-ab12" + +printf 'from-cd23\n' > f.txt +record_change "cd23 edits inherited f.txt" >/dev/null +assert_file_content "cd23 records against inherited file" f.txt "from-cd23" + +switch_view ab12 >/dev/null +assert_file_content "parent remains at its own visible generation" f.txt "from-ab12" +switch_view cd23 >/dev/null +assert_file_content "child restores its edit on the same inherited file" f.txt "from-cd23" + +# ── Case 2: selecting ambient graph data does not rewrite TREE history ────── +begin_section "Cross-view insertion is closure-only and idempotent" +make_temp_repo ambient-inode-insert +init_repo + +new_view feature --draft --parent dev --switch >/dev/null +printf 'ambient content\n' > ambient.txt +add_files ambient.txt >/dev/null +record_change "feature adds ambient.txt" >/dev/null + +JOURNAL_BEFORE="$(journal_fingerprint)" +switch_view dev >/dev/null +insert_from_view feature dev >/dev/null +JOURNAL_AFTER_FIRST="$(journal_fingerprint)" +assert_equal_value "first insert does not append snapshot-derived TREE ops" \ + "$JOURNAL_BEFORE" "$JOURNAL_AFTER_FIRST" +assert_file_content "dev materializes selected ambient file" ambient.txt "ambient content" + +insert_from_view feature dev >/dev/null +JOURNAL_AFTER_SECOND="$(journal_fingerprint)" +assert_equal_value "repeated insert leaves TREE lifecycle metadata unchanged" \ + "$JOURNAL_AFTER_FIRST" "$JOURNAL_AFTER_SECOND" +assert_file_content "repeated insert preserves content" ambient.txt "ambient content" +assert_clean "repeated ambient insert leaves dev clean" + +# ── Case 3: siblings create distinct identities and conflict honestly ─────── +begin_section "Sibling same-path creates surface an identity conflict" +make_temp_repo ambient-inode-siblings +init_repo + +new_view bill --draft --parent dev --switch >/dev/null +printf 'from-bill\n' > f.txt +add_files f.txt >/dev/null +record_change "bill creates f.txt" >/dev/null + +switch_view dev >/dev/null +new_view sally --draft --parent dev --switch >/dev/null +printf 'from-sally\n' > f.txt +add_files f.txt >/dev/null +record_change "sally creates f.txt" >/dev/null + +JOURNAL_BEFORE_CONFLICT="$(journal_fingerprint)" +insert_from_view bill sally >/dev/null 2>&1 || true +JOURNAL_AFTER_CONFLICT="$(journal_fingerprint)" +assert_equal_value "combining siblings does not invent TREE lifecycle ops" \ + "$JOURNAL_BEFORE_CONFLICT" "$JOURNAL_AFTER_CONFLICT" +assert_markers "same-path sibling identities surface conflict markers" f.txt +assert_present "Bill's identity content is preserved" f.txt "from-bill" +assert_present "Sally's identity content is preserved" f.txt "from-sally" +assert_occurrences "Bill's content appears once" f.txt "from-bill" 1 +assert_occurrences "Sally's content appears once" f.txt "from-sally" 1 +assert_honest "same-path identity conflict has an honest exit state" f.txt + +CONFLICT_SNAPSHOT="$(snapshot_file f.txt)" +switch_view bill >/dev/null +assert_file_content "Bill's sibling view retains its own file" f.txt "from-bill" +switch_view sally >/dev/null +assert_file_stable "combined sibling conflict survives view switching" \ + f.txt "$CONFLICT_SNAPSHOT" + +# ── Case 4: renaming one identity must preserve the other path owner ───────── +begin_section "Rename one same-path identity without moving its sibling" +make_temp_repo ambient-inode-rename-side +init_repo + +new_view bill --draft --parent dev --switch >/dev/null +printf 'bill body\n' > f.txt +add_files f.txt >/dev/null +record_change "bill creates f.txt" >/dev/null + +switch_view dev >/dev/null +new_view sally --draft --parent dev --switch >/dev/null +printf 'sally body\n' > f.txt +add_files f.txt >/dev/null +record_change "sally creates f.txt" >/dev/null +insert_from_view bill sally >/dev/null 2>&1 || true +assert_markers "rename setup has a same-path conflict" f.txt + +switch_view bill >/dev/null +mv f.txt bill.txt +record_change "bill renames f.txt to bill.txt" >/dev/null + +switch_view sally >/dev/null +insert_from_view bill sally >/dev/null 2>&1 || true +assert_file_content "renamed Bill inode materializes at bill.txt" bill.txt "bill body" +assert_file_content "Sally inode remains at f.txt" f.txt "sally body" +assert_no_markers "separate paths resolve the prior name conflict" f.txt + +switch_view bill >/dev/null +assert_file_content "Bill rename survives a switch" bill.txt "bill body" +assert_file_not_exists "Bill no longer has f.txt" f.txt +switch_view sally >/dev/null +assert_file_content "Sally path survives a switch" f.txt "sally body" +assert_file_content "combined view retains Bill's renamed inode" bill.txt "bill body" + +# ── Case 5: deleting one identity must preserve the other path owner ───────── +begin_section "Delete one same-path identity without deleting its sibling" +make_temp_repo ambient-inode-delete-side +init_repo + +new_view bill --draft --parent dev --switch >/dev/null +printf 'bill body\n' > f.txt +add_files f.txt >/dev/null +record_change "bill creates f.txt" >/dev/null + +switch_view dev >/dev/null +new_view sally --draft --parent dev --switch >/dev/null +printf 'sally body\n' > f.txt +add_files f.txt >/dev/null +record_change "sally creates f.txt" >/dev/null +insert_from_view bill sally >/dev/null 2>&1 || true +assert_markers "delete setup has a same-path conflict" f.txt + +switch_view bill >/dev/null +rm f.txt +record_change "bill deletes f.txt" >/dev/null + +switch_view sally >/dev/null +insert_from_view bill sally >/dev/null 2>&1 || true +assert_file_content "deleting Bill inode preserves Sally inode" f.txt "sally body" +assert_no_markers "single surviving identity has no conflict markers" f.txt +assert_clean "resolved delete-vs-create view is clean" + +switch_view bill >/dev/null +assert_file_not_exists "Bill's deleted inode remains absent" f.txt +switch_view sally >/dev/null +assert_file_content "Sally inode survives delete switch round-trip" f.txt "sally body" + +# ── Case 6: insertion order must not change the combined identity conflict ─── +begin_section "Opposite insertion orders converge to the same conflict" +make_temp_repo ambient-inode-order +init_repo + +new_view bill --draft --parent dev --switch >/dev/null +printf 'bill body\n' > f.txt +add_files f.txt >/dev/null +record_change "bill creates f.txt" >/dev/null + +switch_view dev >/dev/null +new_view sally --draft --parent dev --switch >/dev/null +printf 'sally body\n' > f.txt +add_files f.txt >/dev/null +record_change "sally creates f.txt" >/dev/null + +switch_view dev >/dev/null +new_view merge-a --draft --parent dev >/dev/null +new_view merge-b --draft --parent dev >/dev/null + +switch_view merge-a >/dev/null +insert_from_view bill merge-a >/dev/null 2>&1 || true +insert_from_view sally merge-a >/dev/null 2>&1 || true +assert_markers "merge-a surfaces the same-path conflict" f.txt +MERGE_A_SNAPSHOT="$(snapshot_file f.txt)" + +switch_view merge-b >/dev/null +insert_from_view sally merge-b >/dev/null 2>&1 || true +insert_from_view bill merge-b >/dev/null 2>&1 || true +assert_markers "merge-b surfaces the same-path conflict" f.txt +assert_file_stable "opposite insert order yields identical materialization" \ + f.txt "$MERGE_A_SNAPSHOT" +assert_present "opposite-order result retains Bill" f.txt "bill body" +assert_present "opposite-order result retains Sally" f.txt "sally body" +assert_honest "opposite-order conflict has an honest exit state" f.txt + +# ── Case 7: concurrent renames retain both incomparable destinations ───────── +begin_section "Concurrent renames preserve both names for one inode" +make_temp_repo ambient-inode-concurrent-rename +init_repo +printf 'shared body\n' > original.txt +add_files original.txt >/dev/null +record_change "add original" >/dev/null + +new_view left --draft --parent dev --switch >/dev/null +mv original.txt left.txt +record_change "rename original to left" >/dev/null + +switch_view dev >/dev/null +new_view right --draft --parent dev --switch >/dev/null +mv original.txt right.txt +record_change "rename original to right" >/dev/null +insert_from_view left right >/dev/null 2>&1 || true + +assert_file_content "left concurrent destination is preserved" left.txt "shared body" +assert_file_content "right concurrent destination is preserved" right.txt "shared body" +assert_file_not_exists "concurrently superseded original path is absent" original.txt + +switch_view left >/dev/null +assert_file_content "left source view retains only left destination" left.txt "shared body" +assert_file_not_exists "left source does not inherit right destination" right.txt +switch_view right >/dev/null +assert_file_content "combined view restores left destination" left.txt "shared body" +assert_file_content "combined view restores right destination" right.txt "shared body" + +# ── Case 8: sequential renames causally supersede earlier destinations ─────── +begin_section "Sequential renames retain only the causal successor" +make_temp_repo ambient-inode-sequential-rename +init_repo +printf 'chain body\n' > a.txt +add_files a.txt >/dev/null +record_change "add a" >/dev/null +mv a.txt b.txt +record_change "rename a to b" >/dev/null +mv b.txt c.txt +record_change "rename b to c" >/dev/null + +assert_file_content "sequential rename ends at c.txt" c.txt "chain body" +assert_file_not_exists "sequential rename removes a.txt" a.txt +assert_file_not_exists "later rename supersedes b.txt" b.txt + +new_view descendant --draft --parent dev --switch >/dev/null +assert_file_content "descendant inherits final sequential destination" c.txt "chain body" +assert_file_not_exists "descendant does not resurrect intermediate name" b.txt + +print_summary diff --git a/tests/harness/40_switch_transactionality.sh b/tests/harness/40_switch_transactionality.sh new file mode 100755 index 00000000..32bca180 --- /dev/null +++ b/tests/harness/40_switch_transactionality.sh @@ -0,0 +1,81 @@ +#!/usr/bin/env bash +# 40_switch_transactionality.sh — Recover failed view switches through disk materialization. + +HARNESS_DIR="$(cd "$(dirname "$0")" && pwd)" +source "$HARNESS_DIR/helpers.sh" +source "$HARNESS_DIR/merge_helpers.sh" + +begin_section "Failed materialization rolls back the published view" +make_temp_repo switch-transaction +init_repo + +printf 'cache/\n' > .atomicignore +printf 'source view\n' > source.txt +add_files .atomicignore source.txt >/dev/null +record_change "add source file and ignore rules" >/dev/null +mkdir -p cache +printf 'source cache\n' > cache/state.txt +SOURCE_VIEW="$(current_view)" + +new_view target --draft --parent "$SOURCE_VIEW" --switch >/dev/null +mkdir -p nested +printf 'target view\n' > nested/target.txt +add_files nested/target.txt >/dev/null +record_change "add target file" >/dev/null + +switch_view "$SOURCE_VIEW" >/dev/null +assert_file_not_exists "target-only file is absent on source" nested/target.txt +assert_file_content "source file is intact before failed switch" source.txt "source view" +assert_file_content "source ignored workspace is present before failure" cache/state.txt "source cache" + +assert_failure "switch fails after publishing but before materialization" \ + env ATOMIC_TEST_FAIL_SWITCH_AFTER_PUBLISH=1 "$ATOMIC_BIN" view switch target --force + +# The failed command exits with a durable recovery marker. The next command +# opens the repository writable and must roll the whole transition back. +assert_file_exists "failed switch leaves durable recovery marker" \ + .atomic/deferred-tree-alignment.pending +assert_success "next writable open performs switch recovery" atomic add source.txt +assert_equal_value() { + local desc="$1" expected="$2" actual="$3" + if [[ "$expected" == "$actual" ]]; then + _pass "$desc" + else + _fail "$desc" "expected '$expected', got '$actual'" + fi +} +assert_equal_value "reopen restores source as current view" \ + "$SOURCE_VIEW" "$(cat .atomic/current_view)" +assert_file_not_exists "successful recovery clears pending marker" \ + .atomic/deferred-tree-alignment.pending +assert_file_content "source tracked content survives rollback" source.txt "source view" +assert_file_content "source ignored workspace is restored by rollback" cache/state.txt "source cache" +assert_success "switch succeeds after obstruction is removed" \ + atomic view switch target --force +assert_equal_value "target becomes current after successful switch" \ + target "$(cat .atomic/current_view)" +assert_file_content "target file materializes after successful retry" nested/target.txt "target view" +assert_file_content "inherited source file remains present" source.txt "source view" +assert_file_not_exists "successful switch leaves no recovery marker" \ + .atomic/deferred-tree-alignment.pending + +begin_section "Recovery removes a partially materialized target view" +switch_view "$SOURCE_VIEW" >/dev/null +assert_file_not_exists "target-only file is removed before mixed-state test" nested/target.txt + +assert_failure "switch fails after target files are materialized" \ + env ATOMIC_TEST_FAIL_SWITCH_AFTER_MATERIALIZE=1 "$ATOMIC_BIN" view switch target --force +assert_file_exists "partially materialized target file exists before recovery" nested/target.txt +assert_file_exists "post-materialization failure retains recovery marker" \ + .atomic/deferred-tree-alignment.pending + +assert_success "writable reopen recovers mixed working copy" atomic add source.txt +assert_equal_value "mixed-state recovery restores source current view" \ + "$SOURCE_VIEW" "$(cat .atomic/current_view)" +assert_file_not_exists "mixed-state recovery removes target-only tracked file" nested/target.txt +assert_file_content "mixed-state recovery restores source content" source.txt "source view" +assert_file_content "mixed-state recovery restores ignored workspace" cache/state.txt "source cache" +assert_file_not_exists "mixed-state recovery clears pending marker" \ + .atomic/deferred-tree-alignment.pending + +print_summary From 2be89873d64e1509c7e26a996f1f7bfdf76e03ca Mon Sep 17 00:00:00 2001 From: Bradley Hilton Date: Mon, 21 Sep 2026 12:22:18 -0500 Subject: [PATCH 04/22] fix(switch): keep name-conflicted files that are in the target view's inherited state (#199) --- .../src/repository/materialize.rs | 47 +++++++++ atomic-repository/src/repository/tests/mod.rs | 2 +- .../tests/switch_file_loss_tests.rs | 96 +++++++++++++++++++ tests/harness/39_view_switch_name_conflict.sh | 86 +++++++++++++++++ 4 files changed, 230 insertions(+), 1 deletion(-) create mode 100644 atomic-repository/src/repository/tests/switch_file_loss_tests.rs create mode 100755 tests/harness/39_view_switch_name_conflict.sh diff --git a/atomic-repository/src/repository/materialize.rs b/atomic-repository/src/repository/materialize.rs index 03c87320..cf1f2ba2 100644 --- a/atomic-repository/src/repository/materialize.rs +++ b/atomic-repository/src/repository/materialize.rs @@ -268,6 +268,53 @@ impl Repository { } } + // REV_TREE recovery: a view is a FILTER over the global graph — every + // node it exposes already lives in the graph, and TREE is just a + // single-valued bookkeeping index over that graph. When two inodes + // claim the same path (cross-view creates, a materialization + // name-conflict), iter_tree can only expose the one binding — so a + // switch used to classify a path that the target view's filter DOES + // render as "absent from the new view" and silently DELETED it (see + // `switch_file_loss_tests`). Re-insert any path that the filter + // renders: some REV_TREE-claimed inode whose introducing change is + // visible on the view AND alive under that filter — the same + // predicate the materializer's name-conflict detection uses. + { + use atomic_core::pristine::TreeTxnT; + let mut by_path: std::collections::HashMap> = + std::collections::HashMap::new(); + if let Ok(pairs) = txn.iter_rev_tree() { + for (inode, path) in pairs { + by_path.entry(path).or_default().push(inode); + } + } + for (path, inodes) in by_path { + if paths.contains(&path) { + continue; + } + for inode in inodes { + if let Ok(Some(position)) = txn.inode_position(inode) { + if !view_change_ids.contains(&position.change) { + continue; + } + // The filter exposes this path only if the claimed + // inode's content chain is ALIVE under the filter; + // a visible-but-superseded claimant must not keep + // a path the view does not render. + if crate::repository::status::is_file_alive_via_retrieval( + &txn, + inode, + position, + &view_change_ids, + ) { + paths.insert(path); + break; + } + } + } + } + } + Ok(paths) } diff --git a/atomic-repository/src/repository/tests/mod.rs b/atomic-repository/src/repository/tests/mod.rs index c9d93141..3ef5342a 100644 --- a/atomic-repository/src/repository/tests/mod.rs +++ b/atomic-repository/src/repository/tests/mod.rs @@ -19,7 +19,7 @@ mod record_tests; mod rename_tests; mod shadow_lock_tests; mod status_tests; - +mod switch_file_loss_tests; mod tracking_tests; mod verify_tests; mod view_tests; diff --git a/atomic-repository/src/repository/tests/switch_file_loss_tests.rs b/atomic-repository/src/repository/tests/switch_file_loss_tests.rs new file mode 100644 index 00000000..b3f05d66 --- /dev/null +++ b/atomic-repository/src/repository/tests/switch_file_loss_tests.rs @@ -0,0 +1,96 @@ +//! Regression tests for the view-switch file-loss bug. +//! +//! Bug (ANGS-A20 wireup): a materialization **name-conflict** made +//! `visible_file_paths` drop a path that WAS in the target view's inherited +//! state; `switch_view`'s Phase 2 then classified it as "old view only" +//! and **deleted the file from the working tree**. The deletion repeatedly +//! removed source files that the target view inherits (and left a stuck +//! "modified" state that blocked further switches). +//! +//! Root cause: `TREE` is single-valued per path, so when two inodes visibly +//! claim the same path (a name conflict), `visible_file_paths` only exposed +//! the single binding — whose introducing change may not be visible on the +//! target view even though the path IS in the target's inherited state. + +use super::*; +use crate::apply::CrossViewInsertOptions; +use crate::record::RecordOptions; +use atomic_core::change::ChangeHeader; + +fn record_all(repo: &Repository, message: &str) { + let header = ChangeHeader::new(message); + let options = RecordOptions::new() + .with_all(true) + .save_to_store(true) + .apply_after_record(true); + repo.record(header, options).unwrap(); +} + +/// Two views independently create `f.txt` at distinct inodes (so a +/// name-conflict is genuinely recorded on dev), then a child view is +/// created inheriting that conflicted lineage. Switching dev → child must +/// keep the file — it IS in the child's inherited state — and the conflict +/// must be surfaced, not silently turned into a deletion. +#[test] +fn switch_to_inheriting_child_keeps_name_conflicted_file() { + let (temp_dir, mut repo) = create_temp_repo(); + + // Seed unrelated base so the child view has something to inherit. + let seed = temp_dir.path().join("seed.txt"); + std::fs::write(&seed, "seed\n").unwrap(); + repo.add("seed.txt", TrackingOptions::default()).unwrap(); + record_all(&repo, "base"); + + repo.create_view_from("feature", "dev").unwrap(); + + // feature independently creates f.txt with its own first line. + let new_file = temp_dir.path().join("f.txt"); + repo.switch_view("feature").unwrap(); + std::fs::write(&new_file, "from-feature\nbody\n").unwrap(); + repo.add("f.txt", TrackingOptions::default()).unwrap(); + record_all(&repo, "feature creates f.txt"); + + // dev independently creates f.txt with a DIFFERENT first line. + repo.switch_view("dev").unwrap(); + std::fs::write(&new_file, "from-dev\nbody\n").unwrap(); + repo.add("f.txt", TrackingOptions::default()).unwrap(); + record_all(&repo, "dev creates f.txt"); + + // Insert feature → dev: two inodes claim f.txt → real name conflict. + repo.insert_from_view(CrossViewInsertOptions::new("feature", "dev")) + .unwrap(); + repo.materialize().unwrap(); + + let on_disk = std::fs::read_to_string(&new_file).unwrap(); + assert!( + on_disk.contains(">>>>>>>"), + "precondition: expected name-conflict markers on disk, got:\n{on_disk}" + ); + assert!( + !repo.list_conflicts().unwrap().is_empty(), + "precondition: conflict must be persisted on dev" + ); + + // Child view inheriting the conflicted dev lineage. + repo.create_view_from("child", "dev").unwrap(); + + // Switch dev → child. THE BUG: this deleted the file. + repo.switch_view("child").unwrap(); + + assert!( + new_file.exists(), + "switch deleted a name-conflicted file that is in the child's \ + inherited state" + ); + + // The conflict must be KEPT and REPORTED on the child, not silently + // materialized as a clean single-version file. (Per-view persistence of + // the CONFLICTS table entry is a separate concern; the switch contract + // is: keep the file and keep its conflict markers.) + let on_disk = std::fs::read_to_string(&new_file).unwrap(); + assert!( + on_disk.contains(">>>>>>>"), + "name-conflicted file must keep its conflict markers on the child, \ + got:\n{on_disk}" + ); +} diff --git a/tests/harness/39_view_switch_name_conflict.sh b/tests/harness/39_view_switch_name_conflict.sh new file mode 100755 index 00000000..84e97329 --- /dev/null +++ b/tests/harness/39_view_switch_name_conflict.sh @@ -0,0 +1,86 @@ +#!/usr/bin/env bash +# 39_view_switch_name_conflict.sh +# +# Regression: a view switch must NOT delete a file that has a materialization +# name-conflict when that file IS in the target view's inherited state. +# +# User-facing flow of the bug (ANGS-A20): +# 1. dev and feature independently create the same path with different +# first lines (two inodes claim f.txt) +# 2. insert feature → dev ⇒ a name conflict is recorded for f.txt on dev +# 3. a child view is forked from dev (it inherits the conflicted lineage) +# 4. switching dev → child DELETES f.txt from the working tree (bug) — +# the file must be kept, with its conflict markers +# +# NOTE: the child is created via `atomic view create ` (fork from the +# current view). The `--parent` flag creates a parented overlay whose visible +# set is EMPTY — switching into it deletes the entire working tree, which is +# a different (broader) defect not covered here. +# +# Run against the harness binary: ATOMIC_BIN=path/to/atomic ./39_view_switch_name_conflict.sh +# Pre-fix builds fail at "f.txt survives the switch". + +HARNESS_DIR="$(cd "$(dirname "$0")" && pwd)" +source "$HARNESS_DIR/helpers.sh" + +# ─────────────────────────────────────────────────────────────────────────── +begin_section "Setup: base repo with a tracked seed file" +# ─────────────────────────────────────────────────────────────────────────── +make_temp_repo "view-switch-name-conflict" +init_repo + +create_file "seed.txt" "seed content" +assert_success "add seed.txt" atomic add seed.txt +record_change "seed base" >/dev/null 2>&1 || true + +# ─────────────────────────────────────────────────────────────────────────── +begin_section "Repro: two views claim f.txt independently (name conflict)" +# ─────────────────────────────────────────────────────────────────────────── +new_view "feature" >/dev/null 2>&1 || true +switch_view "feature" >/dev/null 2>&1 || true + +create_file "f.txt" "from-feature first line\nbody line" +assert_success "add f.txt on feature" atomic add f.txt +record_change "feature creates f.txt" >/dev/null 2>&1 || true + +switch_view "dev" >/dev/null 2>&1 || true +create_file "f.txt" "from-dev first line\nbody line" +assert_success "add f.txt on dev" atomic add f.txt +record_change "dev creates f.txt" >/dev/null 2>&1 || true + +# Insert feature → dev: two inodes claim f.txt ⇒ name conflict. +insert_from_view "feature" "dev" >/dev/null 2>&1 || true + +# The conflict must actually be materialized: markers on disk. +if grep -q ">>>>>>>" f.txt 2>/dev/null; then + _pass "name-conflict markers are on disk after insert" +else + _fail "name-conflict markers are on disk after insert" \ + "expected '>>>>>>>' in f.txt, got: $(cat f.txt 2>/dev/null)" +fi + +# ─────────────────────────────────────────────────────────────────────────── +begin_section "Bug: switching to an inheriting child deletes the file" +# ─────────────────────────────────────────────────────────────────────────── +# Fork from the current view (dev): the child inherits dev's lineage. +new_view "child" >/dev/null 2>&1 || true +switch_view "child" >/dev/null 2>&1 || true +assert_current_view "now on child" "child" + +# THE REGRESSION: the file is in the child's inherited state — it must NOT +# be deleted by the switch. +assert_file_exists "f.txt survives the switch (inherited state)" "f.txt" + +# And it must still be the conflicted content, not a silent clean version. +if grep -q ">>>>>>>" f.txt 2>/dev/null; then + _pass "conflict markers are intact after the switch" +else + _fail "conflict markers are intact after the switch" \ + "expected '>>>>>>>' in f.txt after switch, got: $(cat f.txt 2>/dev/null)" +fi + +# The inherited seed file must also survive (sanity: no collateral deletion). +assert_file_exists "seed.txt survives the switch (inherited state)" "seed.txt" + +# ─────────────────────────────────────────────────────────────────────────── +print_summary \ No newline at end of file From a0c1dea82bfef9c52c25693aaca58318118eab4a Mon Sep 17 00:00:00 2001 From: Aaron Ogle Date: Mon, 21 Sep 2026 14:56:51 -0500 Subject: [PATCH 05/22] fix(agent): record session-touched deletions after plugin claim loss (#201) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The opencode plugin attributes working-copy changes via in-memory ownership claims and sends them as record_files manifests. A plugin restart drops every claim; deletions of files the session itself recorded earlier then never appear in any later manifest and strand as pending deletions forever (observed as 16 pending deletions after an ownership lockout forced a plugin restart mid-session). record_turn now augments explicit record_files manifests with Deleted status entries for paths in the session's persisted files_touched before scoping and validation. Attribution stays conservative: files the session never recorded remain out of scope. fix(agent): re-negotiate recording mode when the plugin changes it The plugin declares its recording mode (recording_scope) on every session-start, but an existing session short-circuited before the declaration was read: a session that opted into explicit-files under an older plugin kept explicit_record_files=true forever. Once a newer plugin stops sending record_files, every Stop fails with 'explicit record_files manifest required' and the session's view silently stays empty — observed live when a session resumed after the plugin switched to whole-tree recording. Apply the declared scope on BOTH session-start paths (fresh and re-entered). An explicit-files session follows a plugin that no longer declares the scope; an unknown declared scope refuses session-start instead of silently recording under the wrong mode. --- atomic-agent/src/integrations/install.rs | 4 +- atomic-agent/src/record/mod.rs | 46 +++++++++++--- atomic-agent/src/record/tests.rs | 55 ++++++++++++++++ .../src/turn/orchestrator/session_start.rs | 63 ++++++++++++++++--- atomic-agent/src/turn/orchestrator/tests.rs | 46 ++++++++++++++ 5 files changed, 192 insertions(+), 22 deletions(-) diff --git a/atomic-agent/src/integrations/install.rs b/atomic-agent/src/integrations/install.rs index 00d6d8d4..a025f4d5 100644 --- a/atomic-agent/src/integrations/install.rs +++ b/atomic-agent/src/integrations/install.rs @@ -1097,8 +1097,7 @@ dst = "{0}/skills/atomic-vault/SKILL.md" let pkg = tempfile::tempdir().unwrap(); std::fs::write( pkg.path().join(MANIFEST_FILE), - format!( - r#" + r#" schema = 1 agent = "repo-file-test" version = "1.0.0" @@ -1110,7 +1109,6 @@ package = "atomic-skills" src = "AGENTS.md" dst = "AGENTS.md" "#, - ), ) .unwrap(); diff --git a/atomic-agent/src/record/mod.rs b/atomic-agent/src/record/mod.rs index cc20f4cf..45f01373 100644 --- a/atomic-agent/src/record/mod.rs +++ b/atomic-agent/src/record/mod.rs @@ -139,16 +139,7 @@ pub fn record_turn( repo_root: &Path, options: &TurnRecordOptions<'_>, ) -> AgentResult { - let manifest = scope::manifest(options)?; - if let Some(files) = &manifest { - scope::validate(repo_root, files, options)?; - if files.is_empty() { - return Err(AgentError::EmptyTurn { - session_id: options.session.session_id.clone(), - turn_number: options.turn_number, - }); - } - } + let mut manifest = scope::manifest(options)?; // Step 1: Open the repository read-only for the initial status check. // This can coexist with other readers. Wait for a transient incompatible // writer before deciding whether work or untracked files exist. @@ -180,6 +171,41 @@ pub fn record_turn( reason: format!("Failed to get repository status: {}", e), })?; + // Restart-proof the deletion scope. The plugin's ownership claims live in + // its process memory; a plugin restart mid-session (e.g. after an + // ambiguity lockout) silently drops every claim, and deletions of files + // this session recorded earlier then sit unrecorded forever — no later + // manifest will ever mention them again. Those deletions are still + // attributable from the persisted session state: `files_touched` holds + // exactly the paths this session's own changes introduced or modified. + // Augment the manifest with their Deleted status entries so a lost claim + // cannot strand a deletion. Files the session never recorded stay + // unclaimed, keeping foreign edits out of scope. + if let Some(files) = manifest.as_mut() { + let touched: std::collections::HashSet<&str> = options + .session + .files_touched + .iter() + .map(String::as_str) + .collect(); + for entry in status.entries() { + if entry.status() != atomic_repository::status::FileStatus::Deleted { + continue; + } + let path = entry.path().to_string_lossy().to_string(); + if !files.contains_key(&path) && touched.contains(path.as_str()) { + files.insert(path.clone(), None); + } + } + scope::validate(repo_root, files, options)?; + if files.is_empty() { + return Err(AgentError::EmptyTurn { + session_id: options.session.session_id.clone(), + turn_number: options.turn_number, + }); + } + } + let status = scope::filter(status, manifest.as_ref()); // Check if there's anything to record at all diff --git a/atomic-agent/src/record/tests.rs b/atomic-agent/src/record/tests.rs index 4da15da1..fe58cca1 100644 --- a/atomic-agent/src/record/tests.rs +++ b/atomic-agent/src/record/tests.rs @@ -1245,6 +1245,61 @@ fn scoped_record_empty_missing_invalid_or_stale_manifest_never_sweeps() { ); } +#[test] +fn scoped_record_recovers_session_touched_deletions_after_claim_loss() { + let dir = tempfile::tempdir().unwrap(); + let repo = atomic_repository::Repository::init(dir.path()).unwrap(); + let mut session = make_session(); + session.view_name = repo.current_view().to_string(); + session.explicit_record_files = true; + drop(repo); + std::fs::write(dir.path().join("mine.txt"), "recorded by this session").unwrap(); + std::fs::write( + dir.path().join("foreign.txt"), + "recorded by another session", + ) + .unwrap(); + + // This session records mine.txt; the recorded paths land in the + // persisted session state (files_touched), like the orchestrator does. + let mine_event = + make_event().with_raw_json(serde_json::json!({"record_files":{"mine.txt":scope::fingerprint(dir.path(),"mine.txt").unwrap()}})); + let outcome = record_turn(dir.path(), &make_options(&session, &mine_event)).unwrap(); + session.add_files_touched(outcome.recorded_file_list()); + + // Another session records foreign.txt. + let mut other = AgentSession::new("other-session", "opencode", "OpenCode"); + other.view_name = session.view_name.clone(); + other.explicit_record_files = true; + let foreign_event = make_event().with_raw_json( + serde_json::json!({"record_files":{"foreign.txt":scope::fingerprint(dir.path(),"foreign.txt").unwrap()}}), + ); + record_turn(dir.path(), &make_options(&other, &foreign_event)).unwrap(); + + // Both files are deleted on disk. The plugin restarted, so the stop + // manifest arrives EMPTY — every ownership claim was lost. + std::fs::remove_file(dir.path().join("mine.txt")).unwrap(); + std::fs::remove_file(dir.path().join("foreign.txt")).unwrap(); + let empty = make_event().with_raw_json(serde_json::json!({"record_files":{}})); + + // mine.txt is still attributable from the session state and records as a + // deletion instead of stranding; foreign.txt stays out of scope. + let outcome = record_turn(dir.path(), &make_options(&session, &empty)).unwrap(); + assert_eq!(outcome.recorded_file_list(), &["mine.txt".to_string()]); + + let repo = atomic_repository::Repository::open_existing(dir.path()).unwrap(); + let status = repo + .status(atomic_repository::status::StatusOptions::default().with_untracked(true)) + .unwrap(); + let pending: Vec = status + .entries() + .iter() + .map(|e| e.path().to_string_lossy().to_string()) + .collect(); + assert!(!pending.contains(&"mine.txt".to_string())); + assert!(pending.contains(&"foreign.txt".to_string())); +} + #[test] fn scoped_snapshot_includes_requested_files_restored_to_clean_and_deletions() { let dir = tempfile::tempdir().unwrap(); diff --git a/atomic-agent/src/turn/orchestrator/session_start.rs b/atomic-agent/src/turn/orchestrator/session_start.rs index 1965c2ed..bf795c64 100644 --- a/atomic-agent/src/turn/orchestrator/session_start.rs +++ b/atomic-agent/src/turn/orchestrator/session_start.rs @@ -42,6 +42,12 @@ impl TurnOrchestrator { let mut session = match self.session_store.load(session_id)? { Some(mut existing) => { // Re-entering an existing session (same session_id) + // The plugin re-declares its recording mode on every + // session-start; a session that opted into explicit-files + // under an older plugin must follow a newer plugin that no + // longer declares the scope, or its Stop hooks refuse to + // record forever. + apply_recording_scope(session_id, &event, &mut existing)?; let result = phase::transition( existing.phase, Event::SessionStart, @@ -173,15 +179,15 @@ impl TurnOrchestrator { // Best-effort: if the repo can't be opened or the view already // exists (resumed session), we log and continue — recording will // still work, it just won't have the parent's history. - if event - .raw_json - .as_ref() - .and_then(|v| v.get("recording_scope")) - .and_then(|v| v.as_str()) - == Some("explicit-files-v1") - { - session.explicit_record_files = true; - } + // + // The plugin declares its recording mode on every session-start, so + // the session follows the plugin it is actually talking to. A session + // that opted into explicit-files under an older plugin must + // re-negotiate to whole-tree recording when a newer plugin stops + // declaring the scope — otherwise its Stop hooks refuse to record + // forever ("explicit record_files manifest required") and the + // session's view silently stays empty. + apply_recording_scope(session_id, &event, &mut session)?; // Several OpenCode sessions can share one actual working directory. // Explicit file scopes separate authorship; a common view gives their @@ -551,3 +557,42 @@ impl TurnOrchestrator { } } } + +/// Apply the plugin's declared recording mode to the session. +/// +/// The plugin declares its mode on every session-start, so the session +/// follows the plugin it is actually talking to rather than the one that +/// created it. A session that opted into explicit-files under an older +/// plugin re-negotiates to whole-tree recording when a newer plugin stops +/// declaring the scope; otherwise its Stop hooks refuse to record forever +/// and the session's view silently stays empty. +fn apply_recording_scope( + session_id: &str, + event: &TurnEvent, + session: &mut AgentSession, +) -> AgentResult<()> { + match event + .raw_json + .as_ref() + .and_then(|v| v.get("recording_scope")) + .and_then(|v| v.as_str()) + { + Some("explicit-files-v1") => session.explicit_record_files = true, + Some(other) => { + return Err(crate::error::AgentError::Internal(format!( + "unknown recording_scope '{other}'" + ))) + } + None => { + if session.explicit_record_files { + log::info!( + "Session {} re-negotiated to whole-tree recording: \ + session-start declared no recording_scope", + session_id + ); + session.explicit_record_files = false; + } + } + } + Ok(()) +} diff --git a/atomic-agent/src/turn/orchestrator/tests.rs b/atomic-agent/src/turn/orchestrator/tests.rs index f41d9c5e..6f891935 100644 --- a/atomic-agent/src/turn/orchestrator/tests.rs +++ b/atomic-agent/src/turn/orchestrator/tests.rs @@ -247,6 +247,52 @@ async fn test_session_start_resumes_ended_session() { assert_eq!(session.turn_count, 3); // preserved } +#[tokio::test] +async fn test_session_start_renegotiates_stale_explicit_files_mode() { + let dir = TempDir::new().unwrap(); + let mut orch = make_orchestrator(&dir); + + // A session created under the old explicit-files plugin persists the + // scoped-recording requirement across restarts. + let mut session = AgentSession::new("sess-legacy", "opencode", "OpenCode"); + session.explicit_record_files = true; + orch.session_store.save(&session).unwrap(); + + // A newer plugin declares no recording_scope on session-start; the + // session must follow it instead of refusing every future Stop. + orch.dispatch(session_start_event("sess-legacy")) + .await + .unwrap(); + + let session = orch.session_store.load("sess-legacy").unwrap().unwrap(); + assert!(!session.explicit_record_files); +} + +#[tokio::test] +async fn test_session_start_keeps_explicit_files_when_declared() { + let dir = TempDir::new().unwrap(); + let mut orch = make_orchestrator(&dir); + + let event = TurnEvent::new("sess-scoped", HookType::SessionStart) + .with_raw_json(serde_json::json!({"recording_scope": "explicit-files-v1"})); + orch.dispatch(event).await.unwrap(); + + let session = orch.session_store.load("sess-scoped").unwrap().unwrap(); + assert!(session.explicit_record_files); +} + +#[tokio::test] +async fn test_session_start_refuses_unknown_recording_scope() { + let dir = TempDir::new().unwrap(); + let mut orch = make_orchestrator(&dir); + + let event = TurnEvent::new("sess-unknown", HookType::SessionStart) + .with_raw_json(serde_json::json!({"recording_scope": "explicit-files-v9"})); + assert!(orch.dispatch(event).await.is_err()); + // The refused session must not be persisted in a half-negotiated state. + assert!(orch.session_store.load("sess-unknown").unwrap().is_none()); +} + #[tokio::test] async fn test_session_start_in_sandbox_adopts_view_without_forking() { // Canonical repository with a distinct view the sandbox operates on. From 36fcf33958909522927818a182963e421169e03e Mon Sep 17 00:00:00 2001 From: Bradley Hilton Date: Mon, 21 Sep 2026 17:22:32 -0500 Subject: [PATCH 06/22] fix(record): resolve name conflicts as namespace patches (#203) --- atomic-core/src/change/graph_op.rs | 1 + atomic-core/src/output/repo/mod.rs | 2 + atomic-core/src/output/repo/names.rs | 64 +++++ .../src/record/workflow/globalize/pipeline.rs | 57 +++++ .../src/record/workflow/record/types.rs | 35 ++- .../src/repository/deferred_tree.rs | 63 ++++- atomic-repository/src/repository/insert.rs | 16 +- .../src/repository/materialize.rs | 125 ++++++---- atomic-repository/src/repository/record.rs | 143 ++++++++++- atomic-repository/src/repository/status.rs | 29 +-- .../tests/causal_file_identity_test.rs | 225 ++++++++++++++++++ docs/record-status-name-conflict.md | 114 +++++++++ .../harness/43_record_status_name_conflict.sh | 137 +++++++++++ 13 files changed, 944 insertions(+), 67 deletions(-) create mode 100644 atomic-core/src/output/repo/names.rs create mode 100644 atomic-repository/tests/causal_file_identity_test.rs create mode 100644 docs/record-status-name-conflict.md create mode 100644 tests/harness/43_record_status_name_conflict.sh diff --git a/atomic-core/src/change/graph_op.rs b/atomic-core/src/change/graph_op.rs index 61a43185..b120fc86 100644 --- a/atomic-core/src/change/graph_op.rs +++ b/atomic-core/src/change/graph_op.rs @@ -247,6 +247,7 @@ pub enum GraphOp { /// resolves the conflict by choosing one version. SolveNameConflict { /// The resolution operation + /// (an empty edge list selects `inode` as the surviving path identity). name: EdgeUpdate, /// Path where conflict occurred path: String, diff --git a/atomic-core/src/output/repo/mod.rs b/atomic-core/src/output/repo/mod.rs index 5b6c5634..d1382111 100644 --- a/atomic-core/src/output/repo/mod.rs +++ b/atomic-core/src/output/repo/mod.rs @@ -105,6 +105,7 @@ mod content; mod error; mod file; mod fork; +mod names; mod options; mod outcome; mod repository; @@ -121,6 +122,7 @@ pub use file::{ output_file, output_file_to_buffer, output_file_to_buffer_with_options, FileOutputError, FileOutputOptions, FileOutputResult, }; +pub use names::live_inode_names; pub use options::OutputOptions; pub use outcome::{FileWritten, OutputOutcome}; pub use repository::{ diff --git a/atomic-core/src/output/repo/names.rs b/atomic-core/src/output/repo/names.rs new file mode 100644 index 00000000..05fc5bb4 --- /dev/null +++ b/atomic-core/src/output/repo/names.rs @@ -0,0 +1,64 @@ +//! View-visible namespace bindings, independent of file content aliveness. + +use crate::output::RetrieveOptions; +use crate::pristine::{GraphTxnT, PristineError}; +use crate::types::{ChangePosition, EdgeFlags, GraphNode, NodeId, Position}; + +/// Find live name vertices attached to an inode under a view's change filter. +/// +/// A name resolution or rename can remove a name without deleting the inode's +/// content. Callers can inspect these vertices' bytes to match a filename. +/// +/// # Examples +/// +/// ```rust,ignore +/// let options = RetrieveOptions::new().with_change_filter(visible_changes); +/// let names = live_inode_names(&txn, inode_position, &options)?; +/// ``` +pub fn live_inode_names( + txn: &T, + inode: Position, + options: &RetrieveOptions, +) -> Result>, PristineError> { + let mut names = Vec::new(); + for edge in txn.get_edges(inode.inode_node())? { + let flags = edge.flag(); + if !flags.contains(EdgeFlags::FOLDER | EdgeFlags::PARENT) + || flags.intersects(EdgeFlags::DELETED | EdgeFlags::PSEUDO) + || !options.passes_filter(edge.introduced_by()) + { + continue; + } + let end = edge.dest(); + if end.change.is_root() || end.pos.get() == 0 { + continue; + } + // The predecessor name ends at this position. find_block_end would + // prefer the empty inode marker at the same position for a FileAdd. + let name = txn.find_block(Position::new( + end.change, + ChangePosition::new(end.pos.get() - 1), + ))?; + let mut linked = false; + let mut unlinked = false; + // Namespace edges carry BLOCK|FOLDER together. Inspect those flags + // directly; the content-oriented typed parent iterator can omit them. + for parent in txn.get_edges(name)? { + let flags = parent.flag(); + if flags.contains(EdgeFlags::PARENT | EdgeFlags::FOLDER) + && !flags.contains(EdgeFlags::PSEUDO) + && options.passes_filter(parent.introduced_by()) + { + if flags.contains(EdgeFlags::DELETED) { + unlinked = true; + } else { + linked = true; + } + } + } + if options.passes_filter(name.change) && linked && !unlinked && !names.contains(&name) { + names.push(name); + } + } + Ok(names) +} diff --git a/atomic-core/src/record/workflow/globalize/pipeline.rs b/atomic-core/src/record/workflow/globalize/pipeline.rs index a39af205..97588cd1 100644 --- a/atomic-core/src/record/workflow/globalize/pipeline.rs +++ b/atomic-core/src/record/workflow/globalize/pipeline.rs @@ -328,6 +328,63 @@ where field: "position", })?; + if let Some(retained) = recorded.name_conflict_resolution() { + ctx.add_dependency_by_id(retained.change)?; + ctx.add_dependency_by_id(inode_pos.change)?; + // Selection is explicit; the other operations unlink only the + // conflicting names. A concurrent rename must keep its content. + result.add_hunk(GraphOp::SolveNameConflict { + name: EdgeUpdate { + edges: Vec::new(), + inode: position_to_option_hash_resolved(ctx.txn(), retained, None), + }, + path: path.to_string(), + }); + let mut edges = Vec::new(); + for &name in recorded.name_conflict_bindings() { + ctx.add_dependency_by_id(name.change)?; + for parent in ctx.txn().get_edges(name)? { + let flag = parent.flag(); + if !flag.contains(EdgeFlags::PARENT | EdgeFlags::FOLDER) + || flag.intersects(EdgeFlags::DELETED | EdgeFlags::PSEUDO) + { + continue; + } + ctx.add_dependency_by_id(parent.introduced_by())?; + ctx.add_dependency_by_id(parent.dest().change)?; + let previous = flag - EdgeFlags::PARENT; + edges.push(crate::change::NewEdge { + previous, + flag: previous | EdgeFlags::DELETED, + from: position_to_option_hash_resolved(ctx.txn(), parent.dest(), None), + to: GraphNode { + change: ctx.get_external(name.change), + start: name.start, + end: name.end, + }, + introduced_by: ctx.get_external(parent.introduced_by()), + }); + } + } + if edges.is_empty() { + return Err(GlobalizeError::MissingField { + path: path.to_string(), + field: "live name binding for name-conflict resolution", + }); + } + result.add_hunk(GraphOp::SolveNameConflict { + name: EdgeUpdate { + edges, + inode: position_to_option_hash_resolved(ctx.txn(), inode_pos, None), + }, + path: path.to_string(), + }); + if let Some(ops) = recorded.crdt_ops().cloned() { + result.set_file_ops(ops); + } + return Ok(result); + } + // Track content positions for each hunk to enrich FileOps later let mut hunk_content_ranges: Vec = Vec::new(); diff --git a/atomic-core/src/record/workflow/record/types.rs b/atomic-core/src/record/workflow/record/types.rs index 3a6ce70d..b0041e63 100644 --- a/atomic-core/src/record/workflow/record/types.rs +++ b/atomic-core/src/record/workflow/record/types.rs @@ -9,7 +9,7 @@ use crate::change::{Encoding, FileOps}; use crate::record::workflow::crdt::CrdtBuildStats; use crate::record::workflow::detect::DetectionKind; use crate::record::workflow::graph_op::BuiltHunk; -use crate::types::{Inode, NodeId, Position}; +use crate::types::{GraphNode, Inode, NodeId, Position}; // ============================================================================ // RECORDING STATS @@ -181,6 +181,10 @@ pub struct RecordedFile { /// graph and CRDT detail is not worth the cost. opaque_generated: bool, + /// Select an identity while unlinking competing name vertices. + name_conflict_resolution: Option>, + name_conflict_bindings: Vec>, + /// Pre-globalized graph operations. When set, `assemble_change` uses /// these directly instead of calling `globalize_recorded_file`, which /// avoids re-walking the graph. Produced by the record path when it @@ -218,6 +222,8 @@ impl RecordedFile { crdt_ops: None, crdt_stats: None, opaque_generated: false, + name_conflict_resolution: None, + name_conflict_bindings: Vec::new(), pre_globalized: None, } } @@ -227,6 +233,25 @@ impl RecordedFile { self.old_line_count = Some(count); } + /// Resolve competing names without deleting their inode or content. The + /// bindings must be live name vertices from the recording view's filter. + pub fn set_name_conflict_resolution( + &mut self, + retained: Position, + bindings: Vec>, + ) { + self.name_conflict_resolution = Some(retained); + self.name_conflict_bindings = bindings; + } + + pub fn name_conflict_resolution(&self) -> Option> { + self.name_conflict_resolution + } + + pub fn name_conflict_bindings(&self) -> &[GraphNode] { + &self.name_conflict_bindings + } + /// Get the old (pristine) line count. pub fn old_line_count(&self) -> Option { self.old_line_count @@ -461,11 +486,13 @@ impl RecordedFile { /// Check if empty (no hunks). /// - /// Note: Moved files are never considered empty even if they have no hunks, - /// because the move itself is a meaningful operation that must be recorded. + /// Moves and name resolutions are meaningful namespace operations even + /// when there are no content hunks. #[must_use] pub fn is_empty(&self) -> bool { - if matches!(self.kind, Some(DetectionKind::Moved)) { + if matches!(self.kind, Some(DetectionKind::Moved)) + || self.name_conflict_resolution.is_some() + { return false; } self.hunks.is_empty() diff --git a/atomic-repository/src/repository/deferred_tree.rs b/atomic-repository/src/repository/deferred_tree.rs index 2669bf73..03e1f711 100644 --- a/atomic-repository/src/repository/deferred_tree.rs +++ b/atomic-repository/src/repository/deferred_tree.rs @@ -12,8 +12,16 @@ const DEFERRED_TREE_ALIGNMENT_LOCK: &str = "deferred-tree-alignment.lock"; #[derive(Debug, Clone, PartialEq, Eq, Hash, Serialize, Deserialize)] #[serde(tag = "kind", rename_all = "snake_case")] pub(super) enum DeferredTreeAction { - Set { path: String }, + Set { + path: String, + }, Delete, + /// Unbind `path` from the inode's desired state — a name-conflict + /// resolution that surrenders the name keeps the inode's identity + /// (and its REV_TREE claim) while ceasing to occupy the path. + UnlinkName { + path: String, + }, } #[derive(Debug, Clone, PartialEq, Eq, Hash, Serialize, Deserialize)] @@ -108,6 +116,7 @@ where .filter_map(|(_, op)| match &op.action { DeferredTreeAction::Set { path } => Some(path.clone()), DeferredTreeAction::Delete => None, + DeferredTreeAction::UnlinkName { .. } => None, }) .collect(); desired.insert(inode, DesiredTreePaths { paths }); @@ -334,6 +343,24 @@ pub(super) fn collect_tree_ops( // deferred rename for views where it is visible. push_unique(&mut ops, op); } + GraphOp::SolveNameConflict { name, path } => { + if let Some(inode) = external_position(change_hash, name.inode) { + push_unique( + &mut ops, + DeferredTreeOp { + change: change_hash, + inode, + baseline_path: current_path_for_position(txn, inode)? + .or_else(|| Some(path.clone())), + action: if name.edges.is_empty() { + DeferredTreeAction::Set { path: path.clone() } + } else { + DeferredTreeAction::UnlinkName { path: path.clone() } + }, + }, + ); + } + } GraphOp::FileDel { del, path, .. } | GraphOp::DirDel { del, path } => { if let Some(inode) = external_position(change_hash, del.inode) { push_unique( @@ -386,6 +413,40 @@ pub(super) fn collect_tree_ops( Ok(ops) } +/// Materialize explicit name selections into TREE using stable graph identity. +/// Recording and cross-view insertion share this operation; no content is copied. +pub(super) fn apply_name_selections( + txn: &mut T, + change_id: NodeId, + change: &Change, +) -> Result<(), RepositoryError> { + for op in change.hunks() { + let GraphOp::SolveNameConflict { name, path } = op else { + continue; + }; + if !name.edges.is_empty() { + continue; + } + let unresolved = || RepositoryError::InvalidOperation { + message: format!("cannot resolve retained identity for {path}: name conflict"), + }; + let inode_change = match name.inode.change { + Some(hash) => txn + .get_internal(&hash) + .map_err(|e| RepositoryError::Database(e.to_string()))? + .ok_or_else(unresolved)?, + None => change_id, + }; + let inode = txn + .position_inode(Position::new(inode_change, name.inode.pos)) + .map_err(|e| RepositoryError::Database(e.to_string()))? + .ok_or_else(unresolved)?; + txn.put_tree(path, inode) + .map_err(|e| RepositoryError::Database(e.to_string()))?; + } + Ok(()) +} + impl Repository { fn deferred_tree_journal_path(&self) -> PathBuf { self.dot_dir.join(DEFERRED_TREE_JOURNAL) diff --git a/atomic-repository/src/repository/insert.rs b/atomic-repository/src/repository/insert.rs index 32ea85e5..1882aaae 100644 --- a/atomic-repository/src/repository/insert.rs +++ b/atomic-repository/src/repository/insert.rs @@ -5,7 +5,7 @@ use crate::apply::{ write_change_to_graph, CrossViewInsertOptions, CrossViewInsertOutcome, InsertOptions, InsertOutcome, InsertStats, }; -use crate::repository::deferred_tree::collect_tree_ops; +use crate::repository::deferred_tree::{apply_name_selections, collect_tree_ops}; use atomic_core::change::Insertion; use atomic_core::pristine::InodeGraphOps; use atomic_core::types::{ChangePosition, EdgeFlags, GraphNode, SerializedGraphEdge}; @@ -851,8 +851,12 @@ impl Repository { if let Ok(Some(inode)) = txn.position_inode(inode_pos) { if let Ok(Some(old_path)) = txn.get_path(inode) { + // Only unlink when OUR inode still claims the old + // path — never a sibling inode's claim (a same-path + // name-conflict binds multiple inodes; del_tree_binding + // is inode-scoped per #206). if old_path != *path { - let _ = txn.del_tree(&old_path); + let _ = txn.del_tree_binding(&old_path, inode); } } let _ = txn.put_tree(path, inode); @@ -1816,6 +1820,7 @@ impl Repository { // idempotently. let mut moved_from_disk: Vec = Vec::new(); if !preserve_existing_tree_paths { + apply_name_selections(&mut txn, change_id, &change)?; for graph_op in change.hunks() { if let GraphOp::FileMove { add, path, .. } = graph_op { // add.inode is Position>; resolve to Position. @@ -2278,8 +2283,12 @@ impl Repository { if let Ok(Some(inode)) = txn.position_inode(inode_pos) { if let Ok(Some(old_path)) = txn.get_path(inode) { + // Only unlink when OUR inode still claims the old + // path — never a sibling inode's claim (a same-path + // name-conflict binds multiple inodes; del_tree_binding + // is inode-scoped per #206). if old_path != *path { - let _ = txn.del_tree(&old_path); + let _ = txn.del_tree_binding(&old_path, inode); } } let _ = txn.put_tree(path, inode); @@ -2294,6 +2303,7 @@ impl Repository { // we need to explicitly remove deleted files from the tree tables. // View-aware: only remove if no other view still references the file. if !preserve_existing_tree_paths { + apply_name_selections(&mut txn, change_id, change)?; for deleted_path in outcome.deleted_files() { if let Ok(Some(inode)) = txn.get_inode(deleted_path) { let dominated = is_file_only_on_view(&txn, inode, view_name); diff --git a/atomic-repository/src/repository/materialize.rs b/atomic-repository/src/repository/materialize.rs index cf1f2ba2..3a2916a1 100644 --- a/atomic-repository/src/repository/materialize.rs +++ b/atomic-repository/src/repository/materialize.rs @@ -94,6 +94,85 @@ impl Repository { } } +pub(crate) type NameConflicts = std::collections::HashMap)>>; + +pub(crate) fn live_names_for_path( + txn: &T, + store: &C, + position: Position, + path: &str, + filter: &HashSet, +) -> Result>, RepositoryError> { + let options = atomic_core::output::RetrieveOptions::new().with_change_filter(filter.clone()); + let names = atomic_core::output::repo::live_inode_names(txn, position, &options) + .map_err(|e| RepositoryError::Database(format!("{path}: {e}")))?; + let filename = path.rsplit('/').next().unwrap_or(path).as_bytes(); + let mut matching = Vec::new(); + for name in names { + let mut bytes = vec![0; (name.end.get() - name.start.get()) as usize]; + store + .get_contents(|id| txn.get_external(id).ok().flatten(), name, &mut bytes) + .map_err(|e| RepositoryError::Database(format!("{path}: {e}")))?; + if bytes == filename { + matching.push(name); + } + } + Ok(matching) +} + +/// Find all live identities for same-path creates using the current view's +/// filter. Both record and materialize must see the identities hidden by TREE. +pub(crate) fn collect_name_conflicts( + txn: &atomic_core::pristine::ReadTxn, + store: &C, + paths: &HashSet<&str>, + filter: &HashSet, +) -> Result { + let mut by_path: std::collections::HashMap> = + std::collections::HashMap::new(); + for (inode, path) in txn + .iter_rev_tree() + .map_err(|e| RepositoryError::Database(e.to_string()))? + { + if paths.contains(path.as_str()) { + by_path.entry(path).or_default().push(inode); + } + } + let mut conflicts = std::collections::HashMap::new(); + for (path, candidates) in by_path { + if candidates.len() < 2 { + continue; + } + let mut live = Vec::new(); + for inode in candidates { + let Some(pos) = txn + .inode_position(inode) + .map_err(|e| RepositoryError::Database(format!("{path}: {e}")))? + else { + continue; + }; + if !pos.change.is_root() && !filter.contains(&pos.change) { + continue; + } + if live_names_for_path(txn, store, pos, &path, filter)?.is_empty() { + continue; + } + if super::status::try_is_file_alive_via_retrieval(txn, pos, filter).map_err(|e| { + RepositoryError::Database(format!( + "cannot resolve recorded baseline for {path}: name conflict: {e}" + )) + })? { + live.push((inode, pos)); + } + } + if live.len() >= 2 { + live.sort_by_key(|(inode, pos)| (pos.change.get(), pos.pos.get(), inode.get())); + conflicts.insert(path, live); + } + } + Ok(conflicts) +} + /// Render a name conflict: two or more inodes are alive at the same path on /// this view, so instead of silently emitting whichever inode `TREE` happened /// to keep, wrap every side's materialized content in conflict markers. @@ -536,46 +615,12 @@ impl Repository { // // The (relatively expensive) aliveness probe runs ONLY for paths with // ≥ 2 candidate inodes, so the common single-inode file pays nothing. - let name_conflicts: std::collections::HashMap)>> = { - use atomic_core::pristine::TreeTxnT; - let mut by_path: std::collections::HashMap> = - std::collections::HashMap::new(); - if let Ok(pairs) = txn.iter_rev_tree() { - for (inode, path) in pairs { - by_path.entry(path).or_default().push(inode); - } - } - let filter = change_filter_arc.as_ref(); - let mut conflicts: std::collections::HashMap)>> = - std::collections::HashMap::new(); - for item in &file_items { - let candidates = match by_path.get(&item.path) { - Some(c) if c.len() >= 2 => c, - _ => continue, - }; - let mut live: Vec<(Inode, Position)> = Vec::new(); - for &inode in candidates { - let pos = match txn.inode_position(inode) { - Ok(Some(p)) => p, - _ => continue, - }; - if !pos.change.is_root() && !filter.contains(&pos.change) { - continue; // not visible on this view - } - if crate::repository::status::is_file_alive_via_retrieval( - &txn, inode, pos, filter, - ) { - live.push((inode, pos)); - } - } - if live.len() >= 2 { - // Deterministic order: creating change, then position, then inode. - live.sort_by_key(|(ino, p)| (p.change.get(), p.pos.get(), ino.get())); - conflicts.insert(item.path.clone(), live); - } - } - conflicts - }; + let name_conflicts = collect_name_conflicts( + &txn, + &self.change_store, + &file_items.iter().map(|item| item.path.as_str()).collect(), + &change_filter_arc, + )?; // Phase 3: Create directories needed by passing files let mut result = MaterializeResult::new(); diff --git a/atomic-repository/src/repository/record.rs b/atomic-repository/src/repository/record.rs index a9c9dddf..ffbda3f3 100644 --- a/atomic-repository/src/repository/record.rs +++ b/atomic-repository/src/repository/record.rs @@ -324,6 +324,64 @@ impl Repository { let shared_cached_txn = CachedGraphTxn::new(&shared_txn).map_err(|e| RecordError::Database(e.to_string()))?; + // TREE selects just one inode per path. Recover competing live + // identities exactly as materialization does before diffing that inode. + let selected_paths: HashSet<&str> = files_to_record + .iter() + .filter_map(|e| e.path().to_str()) + .collect(); + let name_conflicts = super::materialize::collect_name_conflicts( + &shared_txn, + &self.change_store, + &selected_paths, + &shared_change_filter, + )?; + + // A namespace conflict has several legitimate identities. TREE's + // occupant is an index artifact, not a choice of which file to keep. + // Prefer an identity whose bytes match the user's resolution. Break + // ties (including newly edited content) by external graph identity, + // never repository-local allocation or cross-view index update order. + let mut selected_name_inodes = std::collections::HashMap::new(); + for entry in &files_to_record { + if entry.status() != FileStatus::Modified { + continue; + } + let path = entry.path().to_string_lossy(); + let Some(sides) = name_conflicts.get(path.as_ref()) else { + continue; + }; + let working = std::fs::read(self.root.join(path.as_ref())).map_err(|e| { + RecordError::Database(format!("cannot resolve name conflict for {path}: {e}")) + })?; + let mut candidates = Vec::new(); + for &(inode, position) in sides { + let options = atomic_core::output::RetrieveOptions::new() + .with_change_filter(shared_change_filter.clone()); + let (content, _) = retrieve_content_with_filter_fast_with_fork_info( + &shared_cached_txn, + &self.change_store, + inode, + position, + options, + ) + .map_err(|e| { + RecordError::Database(format!("cannot resolve name conflict for {path}: {e}")) + })?; + let hash = shared_txn + .get_external(position.change) + .map_err(|e| RecordError::Database(e.to_string()))? + .ok_or_else(|| { + RecordError::Database(format!("{path}: missing identity hash")) + })?; + candidates.push(((content != working, hash, position.pos), (inode, position))); + } + candidates.sort_by_key(|(key, _)| *key); + if let Some((_, identity)) = candidates.first() { + selected_name_inodes.insert(path.into_owned(), *identity); + } + } + if trace_record { eprintln!( "[record] change filter: {} visible changes", @@ -630,7 +688,8 @@ impl Repository { let par_results: Vec = modified_work .par_iter() .map(|(path, _full_path, _)| { - let (file_inode, file_position) = match get_inode_position(&shared_txn, path) { + let (file_inode, file_position) = match selected_name_inodes.get(path).copied() + .map(Ok).unwrap_or_else(|| get_inode_position(&shared_txn, path)) { Ok(v) => v, Err(e) => return ModifiedResult::Error(path.clone(), e), }; @@ -771,6 +830,70 @@ impl Repository { // Merge parallel results back into sequential state for result in par_results { + let resolved_path = match &result { + ModifiedResult::Recorded(path, _) | ModifiedResult::Skipped(path) => Some(path), + ModifiedResult::Error(path, msg) if name_conflicts.contains_key(path) => { + return Err(RecordError::Database(format!( + "cannot resolve recorded baseline for {path}: name conflict: {msg}" + ))); + } + _ => None, + }; + let mut resolved_name = false; + if let Some(path) = resolved_path.filter(|path| { + // --allow-conflict-markers records literal marker bytes; it + // must not implicitly choose a winner for the name conflict. + !options.get_allow_conflict_markers() + || std::fs::read(self.root.join(path)) + .map(|bytes| { + super::materialize::first_conflict_marker_line(&bytes).is_none() + }) + .unwrap_or(false) + }) { + if let Some(sides) = name_conflicts.get(path) { + use atomic_core::crdt::tables::decode_trunk_id; + use atomic_core::pristine::CrdtTxnT; + let (retained, retained_position) = selected_name_inodes[path]; + if !sides.iter().any(|(inode, _)| *inode == retained) { + return Err(RecordError::Database(format!( + "cannot resolve recorded baseline for {path}: selected identity is not visible in the name conflict" + ))); + } + for &(inode, position) in sides { + if inode == retained { + continue; + } + let bindings = super::materialize::live_names_for_path( + &shared_txn, + &self.change_store, + position, + path, + &shared_change_filter, + )?; + let mut unlinked = RecordedFile::new(path); + unlinked.set_inode(inode); + unlinked.set_position(position); + unlinked.set_name_conflict_resolution(retained_position, bindings); + // Namespace resolution preserves the semantic trunk, + // branches, and tokens. It is not a file deletion. + if let Some(key) = shared_txn + .get_crdt_inode_trunk(inode.get()) + .map_err(|e| RecordError::Database(e.to_string()))? + { + let trunk = decode_trunk_id(&key); + unlinked.set_crdt_ops(atomic_core::change::FileOps::new( + trunk, + path.clone(), + None, + )); + } + stats.hunks_created += 2; // identity selection + namespace unlink + stats.edges_modified += 1; + recorded_files.push(unlinked); + } + resolved_name = true; + } + } match result { ModifiedResult::Recorded(path, recorded) => { stats.files_recorded += 1; @@ -795,8 +918,13 @@ impl Repository { recorded_files.push(*recorded); } ModifiedResult::Skipped(path) => { - skipped_paths.push(path); - stats.files_skipped += 1; + if resolved_name { + recorded_paths.push(path); + stats.files_recorded += 1; + } else { + skipped_paths.push(path); + stats.files_skipped += 1; + } } ModifiedResult::Error(path, msg) => { errors.push((path, msg)); @@ -816,6 +944,15 @@ impl Repository { // Check if we actually recorded anything if recorded_files.is_empty() { + if !errors.is_empty() { + return Err(RecordError::Database( + errors + .iter() + .map(|(path, error)| format!("cannot record {path}: {error}")) + .collect::>() + .join("; "), + )); + } return Err(RecordError::NothingToRecord); } diff --git a/atomic-repository/src/repository/status.rs b/atomic-repository/src/repository/status.rs index b7eab48c..af1dfabe 100644 --- a/atomic-repository/src/repository/status.rs +++ b/atomic-repository/src/repository/status.rs @@ -668,6 +668,14 @@ pub(crate) fn is_file_alive_via_retrieval( position: Position, visible_changes: &HashSet, ) -> bool { + try_is_file_alive_via_retrieval(txn, position, visible_changes).unwrap_or(false) +} + +pub(crate) fn try_is_file_alive_via_retrieval( + txn: &T, + position: Position, + visible_changes: &HashSet, +) -> Result { use atomic_core::output::alive::RetrieveOptions; let inode_node = position.inode_node(); @@ -676,14 +684,7 @@ pub(crate) fn is_file_alive_via_retrieval( // Check forward edges from the inode vertex. If any destination // content vertex is alive (per the full supersession logic), the // file has live content. - let edges = match txn.iter_forward(inode_node, false) { - Ok(edges) => edges, - Err(_) => return false, - }; - - if edges.is_empty() { - return false; - } + let edges = txn.iter_forward(inode_node, false)?; for edge in &edges { // Only consider edges introduced by visible changes @@ -691,18 +692,14 @@ pub(crate) fn is_file_alive_via_retrieval( continue; } // Build the destination vertex from the edge - let dest_vertex = match txn.find_block(edge.dest) { - Ok(v) => v, - Err(_) => continue, - }; + let dest_vertex = txn.find_block(edge.dest)?; // Use the retrieval pipeline's supersession-aware aliveness check - match options.is_vertex_alive(txn, dest_vertex) { - Ok(true) => return true, - _ => continue, + if options.is_vertex_alive(txn, dest_vertex)? { + return Ok(true); } } - false + Ok(false) } /// Normalize a tracked path from the TREE table to a relative PathBuf diff --git a/atomic-repository/tests/causal_file_identity_test.rs b/atomic-repository/tests/causal_file_identity_test.rs new file mode 100644 index 00000000..fa76a33f --- /dev/null +++ b/atomic-repository/tests/causal_file_identity_test.rs @@ -0,0 +1,225 @@ +//! File identities are selected by causal visibility, not ambient path equality. + +use atomic_core::change::GraphOp; +use atomic_core::pristine::{GraphTxnT, MutTxnT, TreeTxnT, ViewScope, ViewTxnT}; +use atomic_core::types::{Hash, Inode, Position}; +use atomic_repository::apply::InsertOptions; +use atomic_repository::{RecordOptions, Repository}; + +fn record(repo: &Repository, message: &str) -> Hash { + let outcome = repo + .record_with_message(message, RecordOptions::default()) + .unwrap(); + assert!(!outcome.has_errors(), "{:?}", outcome.errors()); + *outcome.hash() +} + +fn identity(repo: &Repository, path: &str) -> (Inode, Position) { + let txn = repo.pristine().read_txn().unwrap(); + let inode = txn.get_inode(path).unwrap().expect("tracked path"); + let pos = txn.inode_position(inode).unwrap().expect("recorded inode"); + let hash = txn.get_external(pos.change).unwrap().unwrap(); + (inode, Position::new(hash, pos.pos)) +} + +fn seed(repo: &Repository) { + std::fs::write(repo.root().join("seed.txt"), "seed\n").unwrap(); + repo.add("seed.txt", Default::default()).unwrap(); + record(repo, "seed"); +} + +#[test] +fn descendant_add_and_edit_preserve_inherited_inode() { + let dir = tempfile::tempdir().unwrap(); + let mut repo = Repository::init(dir.path()).unwrap(); + seed(&repo); + repo.create_view_from("parent", "dev").unwrap(); + repo.switch_view("parent").unwrap(); + std::fs::write(dir.path().join("f.txt"), "ancestor\n").unwrap(); + repo.add("f.txt", Default::default()).unwrap(); + let creator = record(&repo, "create inherited file"); + let original = identity(&repo, "f.txt"); + + // Empty own change set: this must be ordinary ancestor closure, not a + // copied change log or an ambient fallback to a sibling identity. + let mut txn = repo.pristine().write_txn().unwrap(); + let parent = txn.get_view("parent").unwrap().unwrap(); + txn.create_view("child", ViewScope::Draft, Some(parent.id)) + .unwrap(); + txn.commit().unwrap(); + repo.switch_view("child").unwrap(); + assert_eq!(identity(&repo, "f.txt"), original); + std::fs::write(dir.path().join("f.txt"), "descendant\n").unwrap(); + repo.add("f.txt", Default::default()).unwrap(); + let edit = record(&repo, "edit inherited file"); + assert_eq!(identity(&repo, "f.txt"), original); + let change = repo.load_change(&edit).unwrap(); + assert!(change.dependencies().contains(&creator)); + assert!(!change + .hunks() + .iter() + .any(|h| matches!(h, GraphOp::FileAdd { path, .. } if path == "f.txt"))); +} + +#[test] +fn siblings_create_distinct_identities_even_with_identical_bytes() { + let dir = tempfile::tempdir().unwrap(); + let mut repo = Repository::init(dir.path()).unwrap(); + seed(&repo); + repo.create_view_from("left", "dev").unwrap(); + repo.create_view_from("right", "dev").unwrap(); + repo.switch_view("left").unwrap(); + std::fs::write(dir.path().join("f.txt"), "same bytes\n").unwrap(); + repo.add("f.txt", Default::default()).unwrap(); + let left = record(&repo, "left creates"); + let left_identity = identity(&repo, "f.txt"); + repo.switch_view("right").unwrap(); + std::fs::write(dir.path().join("f.txt"), "same bytes\n").unwrap(); + repo.add("f.txt", Default::default()).unwrap(); + record(&repo, "right independently creates"); + let right_identity = identity(&repo, "f.txt"); + assert_ne!(left_identity.0, right_identity.0); + assert_ne!(left_identity.1, right_identity.1); + repo.insert_change_rec(&left, InsertOptions::default()) + .unwrap(); + repo.materialize().unwrap(); + let bytes = std::fs::read_to_string(dir.path().join("f.txt")).unwrap(); + assert!(bytes.contains("(name conflict)"), "{bytes}"); +} + +#[test] +fn inserting_the_same_creation_preserves_its_inode_and_graph_identity() { + let dir = tempfile::tempdir().unwrap(); + let mut repo = Repository::init(dir.path()).unwrap(); + seed(&repo); + repo.create_view_from("left", "dev").unwrap(); + repo.create_view_from("right", "dev").unwrap(); + repo.switch_view("left").unwrap(); + std::fs::write(dir.path().join("f.txt"), "original\n").unwrap(); + repo.add("f.txt", Default::default()).unwrap(); + let creator = record(&repo, "create once"); + let original = identity(&repo, "f.txt"); + repo.switch_view("right").unwrap(); + repo.insert_change_rec(&creator, InsertOptions::default()) + .unwrap(); + repo.materialize().unwrap(); + assert_eq!(identity(&repo, "f.txt"), original); + repo.insert_change_rec(&creator, InsertOptions::default()) + .unwrap(); + assert_eq!(identity(&repo, "f.txt"), original); + assert_eq!( + std::fs::read(dir.path().join("f.txt")).unwrap(), + b"original\n" + ); +} + +#[test] +fn name_resolution_changes_namespace_edges_not_file_content() { + let dir = tempfile::tempdir().unwrap(); + let mut repo = Repository::init(dir.path()).unwrap(); + seed(&repo); + repo.create_view_from("left", "dev").unwrap(); + repo.create_view_from("right", "dev").unwrap(); + repo.switch_view("left").unwrap(); + std::fs::write(dir.path().join("f.txt"), "left\n").unwrap(); + repo.add("f.txt", Default::default()).unwrap(); + let left = record(&repo, "left creates"); + repo.switch_view("right").unwrap(); + std::fs::write(dir.path().join("f.txt"), "right\n").unwrap(); + repo.add("f.txt", Default::default()).unwrap(); + record(&repo, "right creates"); + let retained = identity(&repo, "f.txt"); + repo.insert_change_rec(&left, InsertOptions::default()) + .unwrap(); + repo.materialize().unwrap(); + std::fs::write(dir.path().join("f.txt"), "right\n").unwrap(); + let resolution = record(&repo, "select right name"); + assert_eq!(identity(&repo, "f.txt"), retained); + let change = repo.load_change(&resolution).unwrap(); + let mut namespace_edges = 0; + for op in change.hunks() { + if let GraphOp::SolveNameConflict { name, .. } = op { + for edge in &name.edges { + assert!( + edge.flag.is_folder(), + "name resolution tombstoned a content edge: {edge:?}" + ); + namespace_edges += 1; + } + } + } + assert!( + namespace_edges > 0, + "resolution must be represented in the canonical graph" + ); + assert!( + !change.file_ops().iter().any(|ops| matches!( + ops.trunk_op(), + Some(atomic_core::crdt::TrunkOp::Delete { .. }) + )), + "removing a name must not delete its semantic trunk" + ); +} + +#[test] +fn namespace_resolution_and_concurrent_rename_preserve_both_identities() { + for rename_first in [false, true] { + let dir = tempfile::tempdir().unwrap(); + let mut repo = Repository::init(dir.path()).unwrap(); + seed(&repo); + repo.create_view_from("left", "dev").unwrap(); + repo.create_view_from("right", "dev").unwrap(); + repo.switch_view("left").unwrap(); + std::fs::write(dir.path().join("f.txt"), "left\n").unwrap(); + repo.add("f.txt", Default::default()).unwrap(); + let left = record(&repo, "left creates"); + let left_identity = identity(&repo, "f.txt"); + repo.switch_view("right").unwrap(); + std::fs::write(dir.path().join("f.txt"), "right\n").unwrap(); + repo.add("f.txt", Default::default()).unwrap(); + record(&repo, "right creates"); + let right_identity = identity(&repo, "f.txt"); + repo.insert_change_rec(&left, InsertOptions::default()) + .unwrap(); + repo.materialize().unwrap(); + + let rename = |repo: &mut Repository| { + repo.switch_view("left").unwrap(); + std::fs::rename(dir.path().join("f.txt"), dir.path().join("saved.txt")).unwrap(); + let hash = record(repo, "concurrent rename"); + assert_eq!(identity(repo, "saved.txt"), left_identity); + repo.switch_view("right").unwrap(); + hash + }; + let mut rename_hash = None; + if rename_first { + rename_hash = Some(rename(&mut repo)); + } + std::fs::write(dir.path().join("f.txt"), "right\n").unwrap(); + record(&repo, "select right name"); + assert_eq!( + identity(&repo, "f.txt"), + right_identity, + "rename_first={rename_first}" + ); + let rename_hash = rename_hash.unwrap_or_else(|| rename(&mut repo)); + repo.insert_change_rec(&rename_hash, InsertOptions::default()) + .unwrap(); + repo.materialize().unwrap(); + assert_eq!(identity(&repo, "f.txt"), right_identity); + assert_eq!(identity(&repo, "saved.txt"), left_identity); + assert_eq!(std::fs::read(dir.path().join("f.txt")).unwrap(), b"right\n"); + assert_eq!( + std::fs::read(dir.path().join("saved.txt")).unwrap(), + b"left\n" + ); + repo.create_view_from("combined", "right").unwrap(); + repo.switch_view("combined").unwrap(); + assert_eq!(identity(&repo, "f.txt"), right_identity); + assert_eq!(identity(&repo, "saved.txt"), left_identity); + assert_eq!( + std::fs::read(dir.path().join("saved.txt")).unwrap(), + b"left\n" + ); + } +} diff --git a/docs/record-status-name-conflict.md b/docs/record-status-name-conflict.md new file mode 100644 index 00000000..f40ea4eb --- /dev/null +++ b/docs/record-status-name-conflict.md @@ -0,0 +1,114 @@ +# Recording a name-conflict resolution + +## Symptom + +After independently creating the same path on two views and inserting both +changes into one view, materialization reports a name conflict. Replacing the +markers with the content of the identity selected by `TREE` leaves `status` +reporting a modification, but `record` previously returned “Nothing to record”. + +The regression is `tests/harness/43_record_status_name_conflict.sh`. It first +reproduced that failure against the unchanged parent branch, before the +implementation was modified. + +## Cause + +`TREE` selects one inode per path; `REV_TREE` can retain multiple independent +identities claiming that path. Materialization examines all live, view-visible +claimants, but recording previously diffed only the selected inode's content. +Choosing that content exactly produces no content edits, while the unresolved +namespace conflict still requires a patch. + +Deferred TREE replay also needs to distinguish an inode's reverse path from +ownership of the forward path. Removing a stale claimant by path alone can +remove another inode's mapping. Changing which claimant TREE selects must +preserve the competing reverse mappings until the graph resolution makes them +inactive for the current view. + +## Causal identity invariants + +- A descendant draft edits the inherited inode, even when its own change log + is empty. Parent-chain visibility is sufficient; no ambient fallback is used. +- Sibling drafts may independently create different identities at the same + path, even with identical bytes. Combining them produces a name conflict. +- Inserting the same creating change preserves its inode and external graph + identity. Repository-local inode numbers are not a cross-repository identity. + +These are asserted directly in +`atomic-repository/tests/causal_file_identity_test.rs`, in addition to the CLI +scenarios in harness 43. + +## Atomic representation + +The resolution is recorded as graph operations, with dependencies on the +creating changes of the identities involved and the name bindings being removed: + +- An empty `SolveNameConflict` edge update identifies the retained inode. TREE + lifecycle replay and insertion interpret it as selecting that identity for + the path, including when the patch is already present in the canonical graph. +- A separate `SolveNameConflict` update tombstones the competing **name's + incoming FOLDER edges**, not its content edges. Its inode, content, semantic + trunk, branches, and tokens remain intact. A concurrent rename can therefore + preserve that identity under a different path. +- Any content edits to the retained identity use the existing edit pipeline. + +Only identities with live names in the recording view's effective filter are +candidates. Restored bytes matching an existing side prefer that identity. +Ambiguous matches or newly edited content use a stable ordering of external +creating-change hash and inode position. The global TREE occupant and local ID +allocation order do not select the winner. Matching bytes are used only when +recording an explicit resolution; they never deduplicate independent creates. + +TREE replay records the removal of a specific path binding, rather than an +unconditional inode deletion, so it does not erase a concurrent rename. Eager +rename handling also checks forward-path ownership before deleting a stale +reverse mapping's path. + +The canonical graph and historical change objects remain available. A sibling +view whose effective filter excludes the resolution still sees its original +content. A child inheriting the resolution sees the resolved identity. The +regression checks both perspectives, dependency membership, restoration, and +round-trip materialization for either original side and for newly edited content. +It also inserts only the resolution patch into a separate view, letting Atomic +bring in its dependency closure, and verifies the resulting materialized file. + +### Additional test-first findings + +The original resolution attempt deleted competing content. A new CLI regression +first demonstrated that resolving the conflict and combining a rename lost the +renamed file. A structural test independently caught non-FOLDER deletion edges +inside `SolveNameConflict`. Both pass after making resolution namespace-only. + +The reverse recording order then exposed selection of the wrong inode through +TREE. The new identity-selection policy and ownership checks pass both orders +while preserving both original identities and their distinct contents. + +Errors resolving the competing graph identities propagate with the path instead +of becoming a clean no-op. Recording errors also surface when no file could be +recorded. + +## Repair workflow + +On the affected view, replace the conflict markers with the intended file +content, then record the path normally: + +```sh +atomic status +atomic record path/to/file -m "Resolve file name conflict" +atomic status +``` + +This works even when the intended content exactly matches the currently selected +inode. Removing and re-adding the file is unnecessary. + +## Verification + +```sh +cargo build -p atomic-cli --release +ATOMIC_BIN="$PWD/target/release/atomic" bash tests/harness/run_all.sh 39 41 +cargo test -p atomic-core -p atomic-repository --lib +cargo test -p atomic-repository --test causal_file_identity_test +``` + +Suite 39 is the unchanged prior view-switch regression. Suite 41 adds the +record/status regression and view-filter checks. diff --git a/tests/harness/43_record_status_name_conflict.sh b/tests/harness/43_record_status_name_conflict.sh new file mode 100644 index 00000000..30465621 --- /dev/null +++ b/tests/harness/43_record_status_name_conflict.sh @@ -0,0 +1,137 @@ +#!/usr/bin/env bash +# 43_record_status_name_conflict.sh +# Regression: resolving a same-path, independent-inode name conflict must not +# leave status saying modified while record silently says the tree is clean. + +HARNESS_DIR="$(cd "$(dirname "$0")" && pwd)" +source "$HARNESS_DIR/helpers.sh" +source "$HARNESS_DIR/merge_helpers.sh" + +for resolution in retained other edited; do + begin_section "Independent file identities produce a materialized name conflict" + make_temp_repo "record-status-name-conflict-$resolution" + init_repo + create_file "seed.txt" $'seed\n' + atomic add seed.txt + record_change "seed base" + new_view "feature" --from dev + new_view "target" --from dev + switch_view "feature" + create_file "f.txt" $'from-feature first line\nbody line\n' + atomic add f.txt + record_change "feature creates f.txt" + switch_view "target" + create_file "f.txt" $'from-target first line\nbody line\n' + atomic add f.txt + record_change "target creates f.txt" + insert_from_view "feature" "target" + assert_output_contains "fixture has a name conflict" "(name conflict)" cat f.txt + + begin_section "Record the clean resolution ($resolution)" + # Keeping TREE's selected side exactly must still produce a patch that + # resolves the competing identity, even though its content diff is empty. + case "$resolution" in + retained) clean=$'from-target first line\nbody line\n' ;; + other) clean=$'from-feature first line\nbody line\n' ;; + edited) clean=$'new resolved first line\nmerged body\n' ;; + esac + create_file "f.txt" "$clean" + assert_status_flag "status detects the unrecorded resolution" "M" "f.txt" + if output=$(record_change "resolve name conflict" f.txt); then + _pass "record accepts the resolution" + else + _fail "record accepts the resolution" "$output" + fi + assert_output_contains "resolution is recorded, not a clean no-op" "resolve name conflict" atomic change + assert_output_contains "patch depends on the competing identity" "feature creates f.txt" atomic change --show-deps + assert_output_contains "patch depends on the retained identity" "target creates f.txt" atomic change --show-deps + assert_status_no_entry "status is clean after recording the resolution" "f.txt" + resolution_hash="$(tip_hash target)" + + begin_section "Resolution survives content retrieval" + create_file "f.txt" $'unrecorded replacement to force retrieval\n' + assert_success "restore the recorded resolution" atomic restore --force + assert_file_content "resolved file survives without conflict markers" "f.txt" "${clean%$'\n'}" + assert_file_content "unrelated seed survives" "seed.txt" "seed" + assert_status_no_entry "restored resolution is clean" "f.txt" + assert_output_not_contains "resolution clears conflict reporting" "f.txt" atomic conflicts + + begin_section "Materialize the resolution on an inheriting view" + new_view replay --from target + switch_view replay + assert_file_content "replayed graph has only the resolved content" "f.txt" "${clean%$'\n'}" + assert_status_no_entry "replayed resolution is clean" "f.txt" + assert_output_not_contains "replayed resolution has no conflict" "f.txt" atomic conflicts + + begin_section "Resolution respects the source view's change filter" + switch_view feature + assert_file_content "competing content survives on its source view" "f.txt" $'from-feature first line\nbody line' + switch_view replay + assert_file_content "resolved content survives the round trip" "f.txt" "${clean%$'\n'}" + assert_status_no_entry "round-trip resolution is clean" "f.txt" + + begin_section "Insert the resolution patch with its dependency closure" + new_view receiver --from dev + switch_view receiver + assert_success "insert just the resolution and its dependencies" atomic insert "$resolution_hash" --deps + assert_file_content "inserted patch materializes the resolved content" "f.txt" "${clean%$'\n'}" + assert_status_no_entry "inserted resolution is clean" "f.txt" + assert_output_not_contains "inserted resolution has no conflict" "f.txt" atomic conflicts +done + +begin_section "Descendant drafts edit the inherited identity" +make_temp_repo "inherited-file-identity" +init_repo +create_file seed.txt $'seed\n' +atomic add seed.txt +record_change "seed" +new_view parent --from dev +switch_view parent +create_file inherited.txt $'ancestor content\n' +atomic add inherited.txt +record_change "ancestor creates inherited.txt" +new_view child --draft --parent parent +switch_view child +assert_file_content "descendant sees inherited file" inherited.txt "ancestor content" +create_file inherited.txt $'descendant edit\n' +atomic add inherited.txt +assert_status_flag "inherited file is modified, not a new file" M inherited.txt +assert_success "record inherited edit" record_change "edit inherited identity" inherited.txt +assert_output_not_contains "inherited edit does not create another identity" '"hunk_type": "FileAdd"' atomic change -f json +assert_output_contains "inherited edit depends on original creation" "ancestor creates inherited.txt" atomic change --show-deps + +begin_section "Namespace resolution must preserve a concurrently renamed identity" +make_temp_repo "resolve-name-versus-rename" +init_repo +create_file seed.txt $'seed\n' +atomic add seed.txt +record_change "seed" +new_view left --from dev +new_view right --from dev +switch_view left +create_file f.txt $'left identity content\n' +atomic add f.txt +record_change "left creates f.txt" +switch_view right +create_file f.txt $'right identity content\n' +atomic add f.txt +record_change "right creates f.txt" +insert_from_view left right +assert_output_contains "two visible identities really conflict" "(name conflict)" cat f.txt +create_file f.txt $'right identity content\n' +assert_success "record namespace selection" record_change "select right identity" f.txt +switch_view left +assert_file_content "left has its original content before rename" f.txt "left identity content" +atomic move f.txt saved.txt +record_change "rename left identity" +rename_hash="$(tip_hash left)" +switch_view right +assert_success "insert concurrent rename" atomic insert "$rename_hash" --deps +assert_file_content "selected identity still owns f.txt" f.txt "right identity content" +assert_file_content "renamed identity retains its content" saved.txt "left identity content" +new_view combined --from right +switch_view combined +assert_file_content "selected identity survives replay" f.txt "right identity content" +assert_file_content "renamed identity survives replay" saved.txt "left identity content" + +print_summary From 156102f92b96a332ba8411dc9285a09c6d0e23ac Mon Sep 17 00:00:00 2001 From: Bradley Hilton Date: Mon, 21 Sep 2026 17:33:06 -0500 Subject: [PATCH 07/22] =?UTF-8?q?fix(view):=20--parent=20creates=20a=20dra?= =?UTF-8?q?ft=20(overlay)=20child=20=E2=80=94=20never=20an=20empty=20share?= =?UTF-8?q?d=20view=20(#200)?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- atomic-cli/src/commands/view/new.rs | 230 +++++++++--------- .../tests/insert_promote_integration_test.rs | 33 ++- .../src/repository/tests/view_tests.rs | 64 +++++ atomic-repository/src/repository/views.rs | 195 +++++++-------- tests/harness/42_view_create_parent.sh | 55 +++++ 5 files changed, 340 insertions(+), 237 deletions(-) create mode 100755 tests/harness/42_view_create_parent.sh diff --git a/atomic-cli/src/commands/view/new.rs b/atomic-cli/src/commands/view/new.rs index 00a00eab..965e7c9f 100644 --- a/atomic-cli/src/commands/view/new.rs +++ b/atomic-cli/src/commands/view/new.rs @@ -137,9 +137,11 @@ fn validate_view_name(name: &str) -> Result<(), String> { /// Create a new view. /// -/// Creates a new view in the repository. By default, the new view starts -/// empty (with no changes inserted). Use `--from` to fork from an existing -/// view, copying all its changes to the new view. +/// A view is a filter over the global graph: every change and node it +/// exposes already lives in the graph. Creation always makes a DRAFT +/// overlay — anchored on a parent — whose change-set membership optionally +/// starts SEEDED from an existing view's membership (`--from`). The new +/// view owns nothing; it selects. #[derive(Parser, Debug, Default)] #[command(name = "create")] pub struct New { @@ -152,12 +154,11 @@ pub struct New { /// Fork from a specific view instead of the current one. /// - /// By default, `view create` forks from the current view. Use - /// `--from ` to fork from a different view instead. - /// - /// The new view inherits all changes from the source and gets - /// its own view filter on the canonical `GRAPH` so that future - /// changes recorded on it are invisible to the source. + /// Seeds the new view's change-set membership from ``: the view + /// exposes the source's nodes through its own filter. Nothing is copied + /// out of the graph — the edges, hunks, and content already live in it — + /// and the source keeps every one of its nodes. Recording on the new + /// view afterwards writes draft edges the source cannot see. #[arg(long, value_name = "VIEW", add = ArgValueCompleter::new(complete_view_names))] pub from: Option, @@ -182,12 +183,15 @@ pub struct New { /// instead of the global graph. When deleted, all their edges are /// cascade-removed with zero orphans. /// - /// Without this flag, views are created as **shared** (permanent). + /// Draft is the ONLY creation scope — every view is born a draft + /// (an overlay filter over the graph) and may be promoted to a Shared + /// root scope with `view promote`. This flag is accepted for clarity + /// and symmetry. /// /// # Examples /// /// ```text - /// # Create a draft feature view parented on dev + /// # Create a draft feature view anchored on dev /// atomic view create feature-auth --draft /// /// # Create a draft workspace with an explicit parent @@ -198,21 +202,21 @@ pub struct New { /// Parent view for the new view. /// - /// Sets the parent in the view hierarchy. The parent determines - /// the overlay chain for graph traversal: a draft workspace sees - /// its own edges plus its parent's effective view (recursively). + /// Anchors the overlay chain: a draft exposes its own change-set plus + /// its parent's effective set (recursively back to the nearest Shared + /// view). Every node it exposes already lives in the graph — a child + /// never copies or owns its parent's nodes. /// - /// Defaults to the current view. Use `--parent` to specify a - /// different parent explicitly. + /// Defaults to the nearest Shared ancestor of the current view. /// /// # Examples /// /// ```text - /// # Parent on a long-lived service view - /// atomic view create feature-login --draft --parent service-auth + /// # Anchor on a long-lived service view + /// atomic view create feature-login --parent service-auth /// - /// # Parent on dev (the default if dev is current) - /// atomic view create bugfix-123 --draft --parent dev + /// # Anchor on dev (the default if dev is current) + /// atomic view create bugfix-123 --parent dev /// ``` #[arg(long, value_name = "VIEW", add = ArgValueCompleter::new(complete_view_names))] pub parent: Option, @@ -249,55 +253,6 @@ impl New { self } - /// Two-tier view creation: --draft and/or --parent - fn run_two_tier(&self, name: &str, repo: &mut Repository) -> CliResult<()> { - use atomic_core::pristine::{MutTxnT, ViewScope, ViewTxnT}; - - let kind = if self.draft { - ViewScope::Draft - } else { - ViewScope::Shared - }; - - // Resolve the parent view name → ID - let parent_name = self - .parent - .clone() - .unwrap_or_else(|| repo.current_view().to_string()); - - let mut txn = repo - .pristine() - .write_txn() - .map_err(|e| CliError::Internal(e.into()))?; - - let parent_view = txn - .get_view(&parent_name) - .map_err(|e| CliError::Internal(e.into()))? - .ok_or_else(|| CliError::ViewNotFound { - name: parent_name.clone(), - })?; - - let parent_id = parent_view.id; - - // Create the view with explicit kind and parent - let _view = txn - .create_view(name, kind, Some(parent_id)) - .map_err(|e| CliError::Internal(e.into()))?; - - txn.commit().map_err(|e| CliError::Internal(e.into()))?; - - let kind_label = if kind.is_draft() { "draft" } else { "shared" }; - - print_success(&format!( - "Created {} view: {} (parent: {})", - kind_label, - style_view(name), - style_view(&parent_name), - )); - - self.maybe_switch(name, repo) - } - /// Optionally switch to the new view and print hint. fn maybe_switch(&self, name: &str, repo: &mut Repository) -> CliResult<()> { if self.switch { @@ -346,76 +301,67 @@ impl Command for New { }); } - // If --draft or --parent is specified, use the two-tier create path - if self.draft || self.parent.is_some() { - return self.run_two_tier(name, &mut repo); - } - - // Determine how to create the new view: - // - // --from X → create Draft parented on X, insert X's changes - // default → create Draft parented on nearest Shared ancestor, - // with an EMPTY change log (no files until `insert`) + // ONE creation concept: a Draft overlay whose filter is anchored on a + // parent and may be SEEDED from a source view's change-set + // membership. The graph holds every node a view exposes; --from only + // selects which EXISTING changes the new view's filter starts with. // - // The new view is a Draft workspace whose edges go to - // GRAPH (filtered by this view's change set). The parent link - // gives the overlay chain read-access to the shared graph for - // record-time diff computation. - // - // When --from is specified, the source's changes are inserted - // immediately so the new view starts with the source's files. - // Without --from, the change log starts empty — the user brings - // in changes explicitly via `insert from-view`. This is the - // normal workflow: - // - // atomic view create feature # empty workspace - // atomic insert from-view dev # inherit dev's files - // # ... make changes, record ... - // atomic insert from-view feature --to-view dev # promote + // --from S → anchor on S, seed from S + // --from S --parent P → anchor on P, seed from S + // --parent P | --draft → anchor on P, empty membership + // (default) → anchor on nearest Shared, empty membership + // (bring files in with + // `atomic insert from-view dev`) if let Some(ref source) = self.from { - // Explicit --from: fork from the specified view. if !repo.view_exists(source).map_err(CliError::Repository)? { return Err(CliError::ViewNotFound { name: source.to_string(), }); } + } - let source_info = repo.get_view_info(source).map_err(CliError::Repository)?; - let change_count = source_info.change_count; + let (anchor, seed) = resolve_overlay_creation(self.from.as_deref(), self.parent.as_deref()); + + let source_info = if let Some(seed_view) = seed.as_deref() { + Some( + repo.get_view_info(seed_view) + .map_err(CliError::Repository)?, + ) + } else { + None + }; - // create_stack_from creates a Draft workspace parented on - // the source, with the source's change log copied over. - repo.create_view_from(name, source) - .map_err(CliError::Repository)?; + repo.create_overlay_view(name, anchor.as_deref(), seed.as_deref()) + .map_err(CliError::Repository)?; - if change_count > 0 { + if let Some(info) = source_info { + if info.change_count > 0 { print_success(&format!( - "Created view: {} (forked from {} with {} changes)", + "Created view: {} (seeded from {} - {} changes)", style_view(name), - style_view(source), - change_count, + style_view(&info.name), + info.change_count, )); } else { print_success(&format!( - "Created view: {} (forked from {} - empty)", + "Created view: {} (seeded from {} - empty)", style_view(name), - style_view(source), + style_view(&info.name), )); } } else { - // No --from: create an empty Draft workspace parented on the - // nearest Shared ancestor. No changes are inherited — the - // user inserts them explicitly. - repo.create_view(name).map_err(CliError::Repository)?; - + let anchored = if let Some(a) = anchor.as_deref() { + a.to_string() + } else { + match repo.nearest_shared_ancestor(repo.current_view()) { + Ok(name) => name, + Err(_) => repo.current_view().to_string(), + } + }; print_success(&format!( - "Created view: {} (forked from {} - empty)", + "Created view: {} (empty workspace, anchored on {})", style_view(name), - style_view( - &repo - .nearest_shared_ancestor(repo.current_view()) - .unwrap_or_else(|_| repo.current_view().to_string()) - ), + style_view(&anchored), )); } @@ -425,6 +371,28 @@ impl Command for New { // Tests +/// Resolve how a new view is anchored and seeded. +/// +/// Returns `(anchor, seed)`: the overlay-chain anchor view (`None` means the +/// repository default — the nearest Shared ancestor of the current view) and +/// the view whose change-set membership seeds the new view's filter. +/// +/// # Model +/// +/// A view is a filter over the global graph: every node it exposes already +/// lives in the graph. Creation selects where the overlay chain anchors and, +/// optionally, which EXISTING changes the new view's filter starts with — +/// nothing is copied out of the graph; no node is ever owned by two views. +pub fn resolve_overlay_creation( + from: Option<&str>, + parent: Option<&str>, +) -> (Option, Option) { + // --from anchors on its source unless an explicit --parent overrides; + // the seed is always the --from source. + let anchor = parent.or(from).map(|s| s.to_string()); + (anchor, from.map(|s| s.to_string())) +} + #[cfg(test)] mod tests { use super::*; @@ -767,4 +735,30 @@ mod tests { other => panic!("Expected ViewNotFound, got: {:?}", other), } } + // ------------------------------------------------------------------------- + // View kind resolution for the two-tier (--draft/--parent) path + // ------------------------------------------------------------------------- + + #[test] + fn overlay_creation_anchors_and_seeds() { + // --from S: anchored on S and seeded from S. + let (anchor, seed) = resolve_overlay_creation(Some("dev"), None); + assert_eq!(anchor, Some("dev".into())); + assert_eq!(seed, Some("dev".into())); + + // --from S --parent P: anchored on P, seeded from S. + let (anchor, seed) = resolve_overlay_creation(Some("dev"), Some("staging")); + assert_eq!(anchor, Some("staging".into())); + assert_eq!(seed, Some("dev".into())); + + // --parent P only: anchored on P, empty membership. + let (anchor, seed) = resolve_overlay_creation(None, Some("staging")); + assert_eq!(anchor, Some("staging".into())); + assert_eq!(seed, None); + + // default: repository chooses the nearest Shared ancestor; no seed. + let (anchor, seed) = resolve_overlay_creation(None, None); + assert_eq!(anchor, None); + assert_eq!(seed, None); + } } diff --git a/atomic-cli/tests/insert_promote_integration_test.rs b/atomic-cli/tests/insert_promote_integration_test.rs index 0f54fb74..ad6b3c80 100644 --- a/atomic-cli/tests/insert_promote_integration_test.rs +++ b/atomic-cli/tests/insert_promote_integration_test.rs @@ -204,11 +204,13 @@ fn insert_change_alias_pick_still_parses() { } #[test] -fn bare_insert_shared_to_shared_requires_confirmation() { +fn parented_view_is_draft_and_bare_insert_needs_no_confirmation() { let dir = repo_with_base(); let root = dir.path(); - // Create a *shared* view parented on dev and switch to it. + // `--parent` creates a Draft (overlay) workspace parented on dev — not a + // broken Shared view — so the view must list as [draft] and a bare insert + // into the shared parent (the safe draft path) needs NO confirmation. assert!( atomic( root, @@ -216,8 +218,15 @@ fn bare_insert_shared_to_shared_requires_confirmation() { ) .status .success(), - "create shared staging" + "create staging (parented on dev)" ); + + let listed = combined(&atomic(root, &["view", "list", "-a"])); + assert!( + listed.contains("staging") && listed.to_lowercase().contains("draft"), + "a --parent view must be a draft (overlay) workspace:\n{listed}" + ); + std::fs::write(root.join("staging.txt"), b"staging work\n").unwrap(); assert!( atomic(root, &["add", "staging.txt"]).status.success(), @@ -230,24 +239,12 @@ fn bare_insert_shared_to_shared_requires_confirmation() { "record staging change" ); - // Non-interactive (piped stdin): without --confirm this must refuse. - let refused = atomic(root, &["insert"]); - let refused_text = combined(&refused); - assert!( - !refused.status.success(), - "shared->shared insert without --confirm should fail:\n{refused_text}" - ); - assert!( - refused_text.to_lowercase().contains("confirm"), - "error should point at --confirm:\n{refused_text}" - ); - - // With --confirm it proceeds in one line. - let ok = atomic(root, &["insert", "--confirm"]); + // Bare insert (draft → shared parent) proceeds without --confirm. + let ok = atomic(root, &["insert"]); let ok_text = combined(&ok); assert!( ok.status.success(), - "insert --confirm should succeed:\n{ok_text}" + "bare insert from a parented draft must not require --confirm:\n{ok_text}" ); assert!( ok_text.contains("Inserted") && ok_text.contains("change"), diff --git a/atomic-repository/src/repository/tests/view_tests.rs b/atomic-repository/src/repository/tests/view_tests.rs index 9143143e..21e0cecd 100644 --- a/atomic-repository/src/repository/tests/view_tests.rs +++ b/atomic-repository/src/repository/tests/view_tests.rs @@ -205,3 +205,67 @@ fn test_view_info_state_methods() { // For an empty view assert!(info.is_empty()); } + +/// The atomic model, operationalized: a view is a FILTER over the graph — +/// `--from` seeds the new draft's change-set membership from the source +/// without the draft owning a single node; the parent chain renders the +/// source's nodes via the overlay. Recording afterwards writes ONLY draft +/// edges (own changes), and the absorbed holder keeps its identity. +#[test] +fn test_create_overlay_view_seeds_membership_without_owning_nodes() { + use crate::record::RecordOptions; + use atomic_core::change::ChangeHeader; + + let (_temp_dir, mut repo) = create_temp_repo(); + + // dev owns one change (base.txt). + let base = _temp_dir.path().join("base.txt"); + std::fs::write(&base, "base\n").unwrap(); + repo.add("base.txt", TrackingOptions::default()).unwrap(); + let header = ChangeHeader::new("base change"); + repo.record( + header.clone(), + RecordOptions::new().with_all(true).save_to_store(true), + ) + .unwrap(); + + // Seed a child overlay from dev: membership is copied, nodes are not. + repo.create_overlay_view("child", None, Some("dev")) + .unwrap(); + let child = repo.get_view_info("child").unwrap(); + assert!(child.scope.is_draft(), "a creation-scoped view is a draft"); + assert_eq!( + child.own_change_count, 0, + "seeding copies MEMBERSHIP, not ownership" + ); + assert_eq!( + child.inherited_change_count, + repo.get_view_info("dev").unwrap().change_count, + "the overlay chain renders the parent's nodes" + ); + + // The child's filter renders dev's file, though the child owns nothing. + let visible = repo.visible_file_paths("child").unwrap(); + assert!( + visible.contains("base.txt"), + "child filter must render the parent's nodes" + ); + + // Recording on the child writes a draft edge — an own change, nothing + // else is duplicated. + repo.switch_view("child").unwrap(); + let own = _temp_dir.path().join("own.txt"); + std::fs::write(&own, "child work\n").unwrap(); + repo.add("own.txt", TrackingOptions::default()).unwrap(); + repo.record( + header, + RecordOptions::new().with_all(true).save_to_store(true), + ) + .unwrap(); + let after = repo.get_view_info("child").unwrap(); + assert_eq!(after.own_change_count, 1, "one draft edge recorded"); + assert!( + repo.get_view_info("dev").unwrap().change_count == child.inherited_change_count, + "the parent held every one of its nodes" + ); +} diff --git a/atomic-repository/src/repository/views.rs b/atomic-repository/src/repository/views.rs index de12b46e..0a8a269f 100644 --- a/atomic-repository/src/repository/views.rs +++ b/atomic-repository/src/repository/views.rs @@ -83,27 +83,48 @@ impl Repository { /// - The view already exists /// - The database operation fails pub fn create_view(&mut self, name: &str) -> Result<(), RepositoryError> { - // Create the workspace directory for this view. - ensure_workspace_dir(&self.dot_dir, name)?; + self.create_overlay_view(name, None, None) + } - // Create a **Draft** view parented on the nearest Shared - // ancestor of the current view. The change log starts EMPTY — - // no changes are inherited automatically. - // - // The parent link gives the view read-access to the shared - // graph content (via the overlay chain) so that `record` can - // compute diffs against the existing state. But no files are - // *materialised* on disk until changes are explicitly inserted - // into this view (which copies them into the view's change log). - // - // This means: - // `view new feature` → empty workspace, no files - // `insert from-view dev feature` → inherits dev's files - // - // Using the nearest Shared ancestor (instead of the current - // view directly) prevents sibling Draft views from seeing - // each other's edges through the overlay chain. - let parent_name = self.nearest_shared_ancestor(&self.current_view.clone())?; + /// Create one view — the ONLY creation concept atomic has. + /// + /// A view is a **filter over the global graph**: every change and node a + /// view exposes already lives in the graph, and the view merely selects + /// which changes it exposes. Creation therefore has exactly two + /// parameters: + /// + /// - `parent_name` — the overlay-chain anchor. Draft views expose their + /// own change-log PLUS the parent's effective set (recursively back to + /// the nearest Shared view). `None` anchors on the nearest Shared + /// ancestor of the current view. + /// - `seed_from` — an optional view whose **change-set membership** is + /// copied into the new view's own log. This seeds the filter: the + /// hunks, edges, and nodes already live in the GRAPH — nothing is + /// copied out of it, and no content materializes until the filter says + /// so. Without a seed the log starts empty (the documented + /// create-then-`insert from-view` workflow). + /// + /// The new view is always a **Draft**. Shared is a *promoted* scope + /// (`view promote`), never a creation flag. + /// + /// `create_view` and `create_view_from` are thin wrappers around this + /// single concept. + pub fn create_overlay_view( + &mut self, + name: &str, + parent_name: Option<&str>, + seed_from: Option<&str>, + ) -> Result<(), RepositoryError> { + // The overlay chain is anchored on the nearest Shared ancestor of the + // current view unless a parent is given. Using the nearest SHARED + // ancestor (instead of the current view directly) prevents sibling + // Draft views from seeing each other's edges through the overlay. + let anchor = match parent_name { + Some(p) => p.to_string(), + None => self.nearest_shared_ancestor(&self.current_view.clone())?, + }; + + ensure_workspace_dir(&self.dot_dir, name)?; let mut txn = self .pristine @@ -120,14 +141,58 @@ impl Repository { }); } - let parent_view = txn - .get_view(&parent_name) + let anchor_view = txn + .get_view(&anchor) .map_err(|e| RepositoryError::Database(e.to_string()))? .ok_or_else(|| RepositoryError::ViewNotFound { - name: parent_name.clone(), + name: anchor.clone(), })?; - txn.create_view(name, ViewScope::Draft, Some(parent_view.id)) + let mut new_view = txn + .create_view(name, ViewScope::Draft, Some(anchor_view.id)) + .map_err(|e| RepositoryError::Database(e.to_string()))?; + + // Seed the filter: copy the source view's change-set MEMBERSHIP into + // the new view's own log. This does NOT re-insert hunks — the edges + // already exist in GRAPH; the new view simply selects them. + if let Some(source_name) = seed_from { + let source_view = txn + .get_view(source_name) + .map_err(|e| RepositoryError::Database(e.to_string()))? + .ok_or_else(|| RepositoryError::ViewNotFound { + name: source_name.to_string(), + })?; + + // Collect membership first so the iterator is dropped before the + // mutable put_change writes below. + let membership: Vec<(NodeId, Hash)> = { + let iter = txn + .iter_changes(&source_view, 0) + .map_err(|e| RepositoryError::Database(e.to_string()))?; + let mut seeded: Vec<(NodeId, Hash)> = Vec::new(); + for item in iter { + let (_seq, node_id, _merkle) = + item.map_err(|e| RepositoryError::Database(e.to_string()))?; + let hash = txn + .get_external(node_id) + .map_err(|e| RepositoryError::Database(e.to_string()))? + .ok_or_else(|| { + RepositoryError::Database(format!( + "Change {} has no external hash", + node_id.0 + )) + })?; + seeded.push((node_id, hash)); + } + seeded + }; + for (node_id, hash) in membership { + txn.put_change(&mut new_view, node_id, &hash) + .map_err(|e| RepositoryError::Database(e.to_string()))?; + } + } + + txn.update_view(&new_view) .map_err(|e| RepositoryError::Database(e.to_string()))?; txn.commit() @@ -326,83 +391,11 @@ impl Repository { /// repo.create_view_from("feature", "dev")?; /// ``` pub fn create_view_from(&mut self, name: &str, from_view: &str) -> Result<(), RepositoryError> { - let mut txn = self - .pristine - .write_txn() - .map_err(|e| RepositoryError::Database(e.to_string()))?; - - // Check if the new view already exists - if txn - .get_view(name) - .map_err(|e| RepositoryError::Database(e.to_string()))? - .is_some() - { - return Err(RepositoryError::ViewAlreadyExists { - name: name.to_string(), - }); - } - - // Get the source view - let source_view = txn - .get_view(from_view) - .map_err(|e| RepositoryError::Database(e.to_string()))? - .ok_or_else(|| RepositoryError::ViewNotFound { - name: from_view.to_string(), - })?; - - let source_id = source_view.id; - - // Collect all changes from the source view - let changes: Vec<(NodeId, Hash)> = { - let iter = txn - .iter_changes(&source_view, 0) - .map_err(|e| RepositoryError::Database(e.to_string()))?; - - let mut result = Vec::new(); - for item in iter { - let (_seq, node_id, _merkle) = - item.map_err(|e| RepositoryError::Database(e.to_string()))?; - let hash = txn - .get_external(node_id) - .map_err(|e| RepositoryError::Database(e.to_string()))? - .ok_or_else(|| { - RepositoryError::Database(format!( - "Change {} has no external hash", - node_id.0 - )) - })?; - result.push((node_id, hash)); - } - result - }; - - // Create the new view as a **Draft** view parented on the - // source view. Draft views write edges to GRAPH like all - // views, but use a change filter for isolation. The parent - // link means the view chain includes the source's content. - // Create workspace directory for the new view. - ensure_workspace_dir(&self.dot_dir, name)?; - - let mut new_view = txn - .create_view(name, ViewScope::Draft, Some(source_id)) - .map_err(|e| RepositoryError::Database(e.to_string()))?; - - // Copy all changes from the source to the new view's log. - // This does NOT re-insert hunks — the edges already exist in - // GRAPH. The new view sees them via the change filter. - for (node_id, hash) in changes { - txn.put_change(&mut new_view, node_id, &hash) - .map_err(|e| RepositoryError::Database(e.to_string()))?; - } - - // Update the view state - txn.update_view(&new_view) - .map_err(|e| RepositoryError::Database(e.to_string()))?; - - txn.commit() - .map_err(|e| RepositoryError::Database(e.to_string()))?; - - Ok(()) + // Thin wrapper: anchor the new draft on the source AND seed its + // change-set membership from the source's visible set. The overlay + // never copies content — the edges already live in the graph; the + // new view just selects them (see `create_overlay_view`). + self.create_overlay_view(name, Some(from_view), Some(from_view)) } /// List all views in the repository. diff --git a/tests/harness/42_view_create_parent.sh b/tests/harness/42_view_create_parent.sh new file mode 100755 index 00000000..0bc29dbf --- /dev/null +++ b/tests/harness/42_view_create_parent.sh @@ -0,0 +1,55 @@ +#!/usr/bin/env bash +# 42_view_create_parent.sh +# +# Regression: `atomic view create --parent

` must create a child +# view that INHERITS the parent's state — not an empty view. Switching into +# the child must therefore keep the parent's files on disk. +# +# Bug (pre-fix): `--parent` without `--draft` created a SHARED view whose +# visible change set is EMPTY (parent-chain visibility only exists for +# draft/overlay views). `view switch` into it then deleted EVERY tracked +# file from the working tree (`old_files - ∅`). +# +# Run against the harness binary: ATOMIC_BIN=path/to/atomic ./42_view_create_parent.sh +# Pre-fix builds fail at "seed.txt survives the switch into --parent child". + +HARNESS_DIR="$(cd "$(dirname "$0")" && pwd)" +source "$HARNESS_DIR/helpers.sh" + +# ─────────────────────────────────────────────────────────────────────────── +begin_section "Setup: base repo with a tracked file on dev" +# ─────────────────────────────────────────────────────────────────────────── +make_temp_repo "view-create-parent" +init_repo + +create_file "seed.txt" "seed content" +assert_success "add seed.txt" atomic add seed.txt +record_change "seed base" >/dev/null 2>&1 || true + +# ─────────────────────────────────────────────────────────────────────────── +begin_section "Bug: --parent child is empty, switch wipes the tree" +# ─────────────────────────────────────────────────────────────────────────── +$ATOMIC_BIN view create child --parent dev >/dev/null 2>&1 || true + +if $ATOMIC_BIN view list -a 2>/dev/null | grep -q "child"; then + _pass "--parent child view is created" +else + _fail "--parent child view is created" "view create failed" +fi + +switch_view "child" >/dev/null 2>&1 || true +assert_current_view "now on child" "child" + +# THE REGRESSION: the child inherited dev's state, so switching must keep +# the tracked files on disk. +assert_file_exists "seed.txt survives the switch into --parent child" "seed.txt" + +# ─────────────────────────────────────────────────────────────────────────── +begin_section "Round-trip: dev still holds the file after switching back" +# ─────────────────────────────────────────────────────────────────────────── +switch_view "dev" >/dev/null 2>&1 || true +assert_current_view "back on dev" "dev" +assert_file_exists "seed.txt is intact back on dev" "seed.txt" + +# ─────────────────────────────────────────────────────────────────────────── +print_summary \ No newline at end of file From fe51d4b803b600db36006444f3d2022a629a66af Mon Sep 17 00:00:00 2001 From: Bradley Hilton Date: Mon, 21 Sep 2026 17:33:39 -0500 Subject: [PATCH 08/22] chore(harness): unique test harness numbering so 39_view_switch becomes 41_view_switch (#208) --- ...witch_name_conflict.sh => 41_view_switch_name_conflict.sh} | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) rename tests/harness/{39_view_switch_name_conflict.sh => 41_view_switch_name_conflict.sh} (98%) diff --git a/tests/harness/39_view_switch_name_conflict.sh b/tests/harness/41_view_switch_name_conflict.sh similarity index 98% rename from tests/harness/39_view_switch_name_conflict.sh rename to tests/harness/41_view_switch_name_conflict.sh index 84e97329..c745f6cd 100755 --- a/tests/harness/39_view_switch_name_conflict.sh +++ b/tests/harness/41_view_switch_name_conflict.sh @@ -1,5 +1,5 @@ #!/usr/bin/env bash -# 39_view_switch_name_conflict.sh +# 41_view_switch_name_conflict.sh # # Regression: a view switch must NOT delete a file that has a materialization # name-conflict when that file IS in the target view's inherited state. @@ -17,7 +17,7 @@ # set is EMPTY — switching into it deletes the entire working tree, which is # a different (broader) defect not covered here. # -# Run against the harness binary: ATOMIC_BIN=path/to/atomic ./39_view_switch_name_conflict.sh +# Run against the harness binary: ATOMIC_BIN=path/to/atomic ./41_view_switch_name_conflict.sh # Pre-fix builds fail at "f.txt survives the switch". HARNESS_DIR="$(cd "$(dirname "$0")" && pwd)" From 069297834d5810e0469074a23812c44d8f1ae1d2 Mon Sep 17 00:00:00 2001 From: Vincent Date: Mon, 21 Sep 2026 19:38:07 -0400 Subject: [PATCH 09/22] fix(cli): support safe unrecord by hash or prefix (#196) --- atomic-cli/src/commands/unrecord.rs | 52 +++- atomic-cli/tests/unrecord_integration_test.rs | 253 ++++++++++++++++++ atomic-repository/src/repository/history.rs | 91 ++++++- .../tests/unrecord_safety_test.rs | 131 +++++++++ 4 files changed, 512 insertions(+), 15 deletions(-) create mode 100644 atomic-cli/tests/unrecord_integration_test.rs create mode 100644 atomic-repository/tests/unrecord_safety_test.rs diff --git a/atomic-cli/src/commands/unrecord.rs b/atomic-cli/src/commands/unrecord.rs index 42e0f691..47bd3df1 100644 --- a/atomic-cli/src/commands/unrecord.rs +++ b/atomic-cli/src/commands/unrecord.rs @@ -40,22 +40,23 @@ use clap::Parser; -use atomic_core::types::Base32; +use atomic_core::types::{Base32, Hash}; use atomic_repository::unrecord::UnrecordOptions; -use atomic_repository::Repository; +use atomic_repository::{Repository, RepositoryError}; use crate::commands::{find_repository_root, Command}; use crate::error::{CliError, CliResult}; use crate::output::{print_success, print_warning}; -/// Remove the last change from the current view. +/// Remove a change from the current view (default: the last change). /// /// The change is removed from the view's change log but NOT deleted /// from the change store. It can be re-inserted later with `atomic insert`. /// -/// This is the inverse of `atomic record` — it "un-records" a change, -/// reverting the view to the state before that change was applied. +/// This is the inverse of `atomic record` — it "un-records" the selected +/// change while retaining the other changes on the view. /// The working copy is NOT modified; files remain on disk as-is. +/// Changes required by other changes in this view cannot be unrecorded. /// /// # Workflow /// @@ -96,14 +97,10 @@ impl Command for Unrecord { UnrecordOptions::new() }; - let outcome = if let Some(ref _prefix) = self.change { - // TODO: support unrecording a specific change by hash prefix - // once hash_from_prefix is available on the transaction trait. - return Err(CliError::InvalidArgument { - message: "Unrecording a specific change by hash is not yet supported. \ - Use `atomic unrecord` (no argument) to unrecord the last change." - .to_string(), - }); + let outcome = if let Some(ref prefix) = self.change { + let hash = resolve_change(&repo, prefix)?; + repo.unrecord(&hash, options) + .map_err(CliError::Repository)? } else { // Unrecord the most recent change repo.unrecord_last(options).map_err(|e| match e { @@ -132,6 +129,35 @@ impl Command for Unrecord { } } +fn resolve_change(repo: &Repository, prefix: &str) -> CliResult { + // Validate input shape first for a precise error — the shared resolver + // treats malformed input as "no match". + let prefix = prefix.to_ascii_uppercase(); + if prefix.is_empty() + || prefix.len() > 52 + || !prefix + .bytes() + .all(|b| b.is_ascii_uppercase() || (b'2'..=b'7').contains(&b)) + { + return Err(CliError::InvalidArgument { + message: "Expected a Base32 change hash or a non-empty unique prefix".into(), + }); + } + + // The repository's shared, case-insensitive prefix resolver (the same + // one `atomic insert` uses) matches against the whole change store, so a + // hash that exists only on another view still resolves here and then + // hits the repository's explicit membership guard in `unrecord`. + match repo.find_change_by_prefix(&prefix) { + Ok(Some(hash)) => Ok(hash), + Ok(None) => Err(CliError::ChangeNotFound { hash: prefix }), + Err(RepositoryError::AmbiguousHash { prefix, matches }) => Err(CliError::AmbiguousHash { + hash: format!("{prefix} (matches: {})", matches.join(", ")), + }), + Err(e) => Err(CliError::Internal(anyhow::anyhow!("{e}"))), + } +} + #[cfg(test)] mod tests { use super::*; diff --git a/atomic-cli/tests/unrecord_integration_test.rs b/atomic-cli/tests/unrecord_integration_test.rs new file mode 100644 index 00000000..2ba929f5 --- /dev/null +++ b/atomic-cli/tests/unrecord_integration_test.rs @@ -0,0 +1,253 @@ +//! Exercise hash selection and safe unrecord through the real CLI. +use std::collections::HashMap; +use std::fs; +use std::path::Path; +use std::process::Command; + +use atomic_core::change::{Author, Change, ChangeHeader}; +use atomic_core::types::{Base32, Hash}; +use atomic_repository::history::HistoryOptions; +use atomic_repository::{RecordOptions, Repository}; +use tempfile::TempDir; + +fn run(root: &Path, args: &[&str], succeeds: bool) -> String { + let output = Command::new(env!("CARGO_BIN_EXE_atomic")) + .args(args) + .current_dir(root) + .output() + .expect("run CLI"); + let text = format!( + "{}{}", + String::from_utf8_lossy(&output.stdout), + String::from_utf8_lossy(&output.stderr) + ); + assert_eq!(output.status.success(), succeeds, "{args:?}: {text}"); + text +} + +fn record(repo: &Repository, message: &str) -> Hash { + *repo + .record( + ChangeHeader::builder() + .message(message) + .author(Author::new("Test", Some("test@example.com"))) + .build(), + RecordOptions::default(), + ) + .unwrap() + .hash() +} + +fn fixture() -> (TempDir, Vec) { + let dir = TempDir::new().unwrap(); + let repo = Repository::init(dir.path()).unwrap(); + let mut hashes = Vec::new(); + for name in ["a.txt", "b.txt", "c.txt"] { + fs::write(dir.path().join(name), name).unwrap(); + repo.add(name, Default::default()).unwrap(); + hashes.push(record(&repo, name)); + } + (dir, hashes) +} + +fn history(repo: &Repository) -> Vec { + repo.log(HistoryOptions::default().include_inherited(true)) + .unwrap() + .into_iter() + .map(|entry| entry.hash) + .collect() +} + +#[test] +fn full_hash_removes_middle_change_preserving_files_store_and_other_view() { + let (dir, hashes) = fixture(); + let sibling; + { + let mut repo = Repository::open(dir.path()).unwrap(); + let source = repo.current_view().to_string(); + repo.create_view_from("sibling", &source).unwrap(); + sibling = repo + .log( + HistoryOptions::default() + .view("sibling") + .include_inherited(true), + ) + .unwrap(); + } + fs::write(dir.path().join("b.txt"), "unrecorded local edits").unwrap(); + run(dir.path(), &["unrecord", &hashes[1].to_base32()], true); + let repo = Repository::open(dir.path()).unwrap(); + assert_eq!(history(&repo), vec![hashes[0], hashes[2]]); + assert!(repo.load_change(&hashes[1]).is_ok()); + assert!(repo + .get_file_content("b.txt") + .unwrap() + .unwrap_or_default() + .is_empty()); + assert_eq!( + repo.get_file_content_on_view("b.txt", "sibling").unwrap(), + Some(b"b.txt".to_vec()) + ); + assert_eq!( + fs::read_to_string(dir.path().join("b.txt")).unwrap(), + "unrecorded local edits" + ); + for name in ["a.txt", "c.txt"] { + assert_eq!(fs::read_to_string(dir.path().join(name)).unwrap(), name); + } + let after = repo + .log( + HistoryOptions::default() + .view("sibling") + .include_inherited(true), + ) + .unwrap(); + assert_eq!( + after.iter().map(|e| e.hash).collect::>(), + sibling.iter().map(|e| e.hash).collect::>() + ); + // The retained change is usable, not just a leftover object on disk. + repo.insert_change(&hashes[1], Default::default()).unwrap(); + assert_eq!( + repo.get_file_content("b.txt").unwrap(), + Some(b"b.txt".to_vec()) + ); +} + +#[test] +fn lowercase_unique_prefix_and_dry_run() { + let (dir, hashes) = fixture(); + let prefix = hashes[1].to_base32()[..16].to_ascii_lowercase(); + let preview = run(dir.path(), &["unrecord", &prefix, "--dry-run"], true); + assert!(preview.contains(&hashes[1].to_base32())); + assert_eq!(history(&Repository::open(dir.path()).unwrap()), hashes); + run(dir.path(), &["unrecord", &prefix], true); + assert_eq!( + history(&Repository::open(dir.path()).unwrap()), + vec![hashes[0], hashes[2]] + ); +} + +#[test] +fn no_argument_still_removes_last_change() { + let (dir, hashes) = fixture(); + run(dir.path(), &["unrecord", "-n"], true); + assert_eq!(history(&Repository::open(dir.path()).unwrap()), hashes); + run(dir.path(), &["unrecord"], true); + assert_eq!(history(&Repository::open(dir.path()).unwrap()), hashes[..2]); +} + +#[test] +fn malformed_unknown_and_nonmember_hashes_do_not_mutate() { + let (dir, hashes) = fixture(); + let absent = Hash::of(b"not in this repository").to_base32(); + for target in ["", "0123456789", "A/B", &"A".repeat(53), &absent] { + run(dir.path(), &["unrecord", target], false); + run(dir.path(), &["unrecord", target, "-n"], false); + assert_eq!(history(&Repository::open(dir.path()).unwrap()), hashes); + } + // The object still exists in storage after removal, but is not a member. + run(dir.path(), &["unrecord", &hashes[1].to_base32()], true); + assert!(run(dir.path(), &["unrecord", &hashes[1].to_base32()], false).contains("not in view")); + assert!(run( + dir.path(), + &["unrecord", &hashes[1].to_base32(), "-n"], + false + ) + .contains("not in view")); + assert_eq!( + history(&Repository::open(dir.path()).unwrap()), + vec![hashes[0], hashes[2]] + ); +} + +#[test] +fn ambiguous_prefix_is_rejected_without_mutation() { + let (dir, hashes) = fixture(); + let prefix = { + let repo = Repository::open(dir.path()).unwrap(); + let mut seen = HashMap::new(); + let mut collision = None; + // 33 distinct hashes guarantee a collision in the first Base32 digit. + for i in 0..33 { + let hash = repo + .save_change(&Change::empty(ChangeHeader::new(format!("stored {i}")))) + .unwrap(); + let first = hash.to_base32()[..1].to_string(); + if seen.insert(first.clone(), hash).is_some() { + collision = Some(first); + break; + } + } + collision.unwrap() + }; + assert!(run(dir.path(), &["unrecord", &prefix], false) + .to_lowercase() + .contains("ambiguous")); + assert!(run(dir.path(), &["unrecord", &prefix, "-n"], false) + .to_lowercase() + .contains("ambiguous")); + assert_eq!(history(&Repository::open(dir.path()).unwrap()), hashes); +} + +#[test] +fn dependent_change_blocks_both_preview_and_execution() { + let (dir, mut hashes) = fixture(); + { + let repo = Repository::open(dir.path()).unwrap(); + fs::write(dir.path().join("a.txt"), "updated\n").unwrap(); + let dependent = record(&repo, "edit a"); + assert!(repo + .load_change(&dependent) + .unwrap() + .dependencies() + .contains(&hashes[0])); + hashes.push(dependent); + } + for dry in [false, true] { + let hash = hashes[0].to_base32(); + let mut args = vec!["unrecord", &hash]; + if dry { + args.push("-n"); + } + assert!(run(dir.path(), &args, false).contains("depends on it")); + assert_eq!(history(&Repository::open(dir.path()).unwrap()), hashes); + assert_eq!( + fs::read_to_string(dir.path().join("a.txt")).unwrap(), + "updated\n" + ); + } + run(dir.path(), &["unrecord", &hashes[3].to_base32()], true); + run(dir.path(), &["unrecord", &hashes[0].to_base32()], true); + assert_eq!( + history(&Repository::open(dir.path()).unwrap()), + hashes[1..3] + ); +} + +#[test] +fn inherited_changes_are_rejected_in_forked_view() { + let (dir, hashes) = fixture(); + { + let mut repo = Repository::open(dir.path()).unwrap(); + let source = repo.current_view().to_string(); + repo.create_view_from("child", &source).unwrap(); + repo.switch_view("child").unwrap(); + } + for dry in [false, true] { + let hash = hashes[2].to_base32(); + let mut args = vec!["unrecord", &hash]; + if dry { + args.push("-n"); + } + assert!(run(dir.path(), &args, false).contains("inherited")); + assert_eq!(history(&Repository::open(dir.path()).unwrap()), hashes); + } +} + +#[test] +fn empty_view_reports_nothing_to_unrecord() { + let dir = TempDir::new().unwrap(); + drop(Repository::init(dir.path()).unwrap()); + assert!(run(dir.path(), &["unrecord"], false).contains("nothing to unrecord")); +} diff --git a/atomic-repository/src/repository/history.rs b/atomic-repository/src/repository/history.rs index b7b08a8f..bd2e8ea6 100644 --- a/atomic-repository/src/repository/history.rs +++ b/atomic-repository/src/repository/history.rs @@ -1,5 +1,7 @@ use std::collections::HashSet; +use atomic_core::pristine::ViewState; + use super::*; impl Repository { @@ -307,8 +309,11 @@ impl Repository { // Get the view let mut view = txn - .open_or_create_view(view_name) - .map_err(|e| RepositoryError::Database(e.to_string()))?; + .get_view(view_name) + .map_err(|e| RepositoryError::Database(e.to_string()))? + .ok_or_else(|| RepositoryError::ViewNotFound { + name: view_name.to_string(), + })?; // Get internal ID let change_id = txn @@ -318,6 +323,10 @@ impl Repository { hash: hash.to_base32(), })?; + // Validate under the same write transaction used for removal. Preview + // and execution must reject the same unsafe operations. + self.check_unrecord_safety(&txn, &view, hash, change_id)?; + // Check if this is a dry run if options.dry_run { // Preview mode - just return what would happen @@ -375,6 +384,84 @@ impl Repository { Ok(outcome) } + fn check_unrecord_safety( + &self, + txn: &T, + view: &ViewState, + hash: &Hash, + change_id: NodeId, + ) -> Result<(), RepositoryError> { + if txn.get_change_seq(view, change_id)?.is_none() { + return Err(RepositoryError::Unrecord(format!( + "Change {} is not in view '{}'", + hash.to_base32(), + view.name + ))); + } + + // Forked views can contain copied references to ancestor changes. + // Removing that copy cannot hide the inherited change, so refuse it. + let mut remaining_ids = collect_view_change_ids(txn, view)?; + let mut ancestor = if view.kind.is_draft() { + view.parent + } else { + None + }; + let mut seen_views = HashSet::from([view.id]); + while let Some(id) = ancestor { + if !seen_views.insert(id) { + return Err(RepositoryError::Unrecord("Cyclic view ancestry".into())); + } + let parent = txn + .get_view_by_id(id)? + .ok_or_else(|| RepositoryError::Unrecord(format!("Missing ancestor view {id}")))?; + if txn.get_change_seq(&parent, change_id)?.is_some() { + return Err(RepositoryError::Unrecord(format!( + "Change {} is inherited from view '{}'; unrecord it there instead", + hash.to_base32(), + parent.name + ))); + } + remaining_ids.extend(collect_view_change_ids(txn, &parent)?); + ancestor = if parent.kind.is_draft() { + parent.parent + } else { + None + }; + } + + let mut pending = Vec::new(); + for id in remaining_ids { + if id != change_id { + let candidate = txn.get_external(id)?.ok_or_else(|| { + RepositoryError::Unrecord(format!("Missing hash for change {id}")) + })?; + pending.push(candidate); + } + } + let mut checked = HashSet::new(); + while let Some(candidate) = pending.pop() { + if !checked.insert(candidate) { + continue; + } + let id = txn.get_internal(&candidate)?; + let dependencies = match id { + Some(id) if txn.is_change_deps_indexed(id)? => txn.get_change_deps(id)?, + // Old repositories can predate the dependency index. Never + // mistake an unindexed change for one with no dependencies. + _ => self.load_change(&candidate)?.dependencies().to_vec(), + }; + if dependencies.contains(hash) { + return Err(RepositoryError::Unrecord(format!( + "Cannot unrecord {}: change {} in the dependency closure of view '{}' depends on it; unrecord dependent changes first", + hash.to_base32(), candidate.to_base32(), view.name + ))); + } + pending.extend(dependencies); + } + Ok(()) + } + /// Unrecord the last change from the current view. /// /// This is a convenience method for unrecording the most recent change. diff --git a/atomic-repository/tests/unrecord_safety_test.rs b/atomic-repository/tests/unrecord_safety_test.rs new file mode 100644 index 00000000..5d069fdc --- /dev/null +++ b/atomic-repository/tests/unrecord_safety_test.rs @@ -0,0 +1,131 @@ +//! Dependency validation uses stored changes when legacy repositories lack +//! an index, and follows dependencies beyond direct view membership. +use atomic_core::change::{Change, ChangeHeader}; +use atomic_core::pristine::{GraphTxnT, MutTxnT, ViewTxnT}; +use atomic_core::types::Hash; +use atomic_repository::history::HistoryOptions; +use atomic_repository::unrecord::UnrecordOptions; +use atomic_repository::Repository; +use tempfile::TempDir; + +fn save(repo: &Repository, message: &str, deps: Vec) -> Hash { + repo.save_change(&Change::new( + ChangeHeader::new(message), + vec![], + vec![], + deps, + )) + .unwrap() +} + +// Deliberately bypass insert_change: this represents a legacy repository +// that has view membership but no dependency index for these changes. +fn add_legacy_member(repo: &Repository, hash: Hash) { + let mut txn = repo.pristine().write_txn().unwrap(); + let mut view = txn.get_view(repo.current_view()).unwrap().unwrap(); + let id = txn.register_change(&hash).unwrap(); + assert!(!txn.is_change_deps_indexed(id).unwrap()); + txn.put_change(&mut view, id, &hash).unwrap(); + txn.update_view(&view).unwrap(); + txn.commit().unwrap(); +} + +fn history(repo: &Repository) -> Vec { + repo.log(HistoryOptions::default()) + .unwrap() + .into_iter() + .map(|e| e.hash) + .collect() +} + +#[test] +fn unindexed_dependent_blocks_removal_and_dry_run() { + let dir = TempDir::new().unwrap(); + let repo = Repository::init(dir.path()).unwrap(); + let a = save(&repo, "a", vec![]); + let b = save(&repo, "b", vec![a]); + add_legacy_member(&repo, a); + add_legacy_member(&repo, b); + for options in [UnrecordOptions::dry_run(), UnrecordOptions::new()] { + let err = repo.unrecord(&a, options).unwrap_err(); + assert!(err.to_string().contains("depends on it"), "{err}"); + assert_eq!(history(&repo), vec![a, b]); + } + repo.unrecord(&b, UnrecordOptions::new()).unwrap(); + repo.unrecord(&a, UnrecordOptions::new()).unwrap(); + assert!(history(&repo).is_empty()); + assert!(repo.has_change(&a)); + assert!(repo.has_change(&b)); +} + +#[test] +fn transitive_dependency_outside_direct_membership_blocks_removal() { + let dir = TempDir::new().unwrap(); + let repo = Repository::init(dir.path()).unwrap(); + let a = save(&repo, "a", vec![]); + let b = save(&repo, "b", vec![a]); + let c = save(&repo, "c", vec![b]); + add_legacy_member(&repo, a); + // b is visible through c's dependency closure, without its own view row. + add_legacy_member(&repo, c); + for options in [UnrecordOptions::dry_run(), UnrecordOptions::new()] { + let err = repo.unrecord(&a, options).unwrap_err(); + assert!(err.to_string().contains("depends on it"), "{err}"); + assert_eq!(history(&repo), vec![a, c]); + } +} + +#[test] +fn unverifiable_dependencies_fail_closed() { + let dir = TempDir::new().unwrap(); + let repo = Repository::init(dir.path()).unwrap(); + let a = save(&repo, "a", vec![]); + let b = save(&repo, "b", vec![Hash::of(b"missing object")]); + add_legacy_member(&repo, a); + add_legacy_member(&repo, b); + for options in [UnrecordOptions::dry_run(), UnrecordOptions::new()] { + assert!(repo.unrecord(&a, options).is_err()); + assert_eq!(history(&repo), vec![a, b]); + } +} + +#[test] +fn inherited_dependent_also_blocks_removal() { + let dir = TempDir::new().unwrap(); + let mut repo = Repository::init(dir.path()).unwrap(); + let parent = repo.current_view().to_string(); + repo.create_view_from("child", &parent).unwrap(); + repo.switch_view("child").unwrap(); + let a = save(&repo, "child change", vec![]); + add_legacy_member(&repo, a); + // The shared parent advances after the draft fork. Its new change is + // inherited, even though it has no row in the child's own history. + repo.switch_view(&parent).unwrap(); + let b = save(&repo, "parent dependent", vec![a]); + add_legacy_member(&repo, b); + repo.switch_view("child").unwrap(); + for options in [UnrecordOptions::dry_run(), UnrecordOptions::new()] { + let err = repo.unrecord(&a, options).unwrap_err(); + assert!(err.to_string().contains("depends on it"), "{err}"); + assert_eq!(history(&repo), vec![a]); + } +} + +#[test] +fn nonexistent_view_is_not_created_by_unrecord() { + let dir = TempDir::new().unwrap(); + let repo = Repository::init(dir.path()).unwrap(); + let hash = save(&repo, "a", vec![]); + add_legacy_member(&repo, hash); + assert!(repo + .unrecord(&hash, UnrecordOptions::new().view("missing")) + .is_err()); + assert!(repo + .pristine() + .read_txn() + .unwrap() + .get_view("missing") + .unwrap() + .is_none()); + assert_eq!(history(&repo), vec![hash]); +} From 7d59890e302365eda4c923a775b93e40931319e5 Mon Sep 17 00:00:00 2001 From: Bradley Hilton Date: Mon, 21 Sep 2026 18:47:56 -0500 Subject: [PATCH 10/22] refactor(change): resolve hash prefixes via the shared repository resolver (#209) Replace change/command.rs's local, case-SENSITIVE prefix scan with Repository::find_change_by_prefix, the same case-insensitive resolver used by 'atomic insert' and 'atomic unrecord'. Removes the last divergent hash-addressing rule in the CLI. --- atomic-cli/src/commands/change/command.rs | 31 ++++++++--------------- 1 file changed, 10 insertions(+), 21 deletions(-) diff --git a/atomic-cli/src/commands/change/command.rs b/atomic-cli/src/commands/change/command.rs index c51ccfba..a3352b7c 100644 --- a/atomic-cli/src/commands/change/command.rs +++ b/atomic-cli/src/commands/change/command.rs @@ -210,33 +210,22 @@ impl ChangeCmd { view_name: &str, prefix: &str, ) -> CliResult<(Hash, Option)> { - // Search for matching changes - let mut matches: Vec = Vec::new(); - - for result in repo.iter_changes() { - let hash = result.map_err(|e| CliError::Internal(anyhow::anyhow!("{}", e)))?; - let hash_str = hash.to_base32(); - if hash_str.starts_with(prefix) { - matches.push(hash); - } - } - - match matches.len() { - 0 => Err(CliError::ChangeNotFound { - hash: prefix.to_string(), - }), - 1 => { - let hash = matches[0]; + // The repository's shared, case-insensitive prefix resolver (the + // same one `atomic insert` and `atomic unrecord` use). + match repo.find_change_by_prefix(prefix) { + Ok(Some(hash)) => { let seq = self.find_sequence_for_hash(repo, view_name, &hash)?; Ok((hash, seq)) } - _ => { - // Format the matches for display in the error message - let match_list: Vec = matches.iter().map(|h| h.to_base32()).collect(); + Ok(None) => Err(CliError::ChangeNotFound { + hash: prefix.to_string(), + }), + Err(atomic_repository::RepositoryError::AmbiguousHash { prefix, matches }) => { Err(CliError::AmbiguousHash { - hash: format!("{} (matches: {})", prefix, match_list.join(", ")), + hash: format!("{prefix} (matches: {})", matches.join(", ")), }) } + Err(e) => Err(CliError::Internal(anyhow::anyhow!("{e}"))), } } From e5d1d3b69d1a85e275f8d506323ec5ec77fabd13 Mon Sep 17 00:00:00 2001 From: "github-actions[bot]" <41898282+github-actions[bot]@users.noreply.github.com> Date: Mon, 21 Sep 2026 23:49:43 +0000 Subject: [PATCH 11/22] Bump version to 0.18.3 --- Cargo.lock | 22 +++++++++++----------- Cargo.toml | 2 +- 2 files changed, 12 insertions(+), 12 deletions(-) diff --git a/Cargo.lock b/Cargo.lock index a4bd3044..830d8ca4 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -129,7 +129,7 @@ dependencies = [ [[package]] name = "atomic-agent" -version = "0.18.2" +version = "0.18.3" dependencies = [ "anyhow", "atomic-canonical", @@ -160,7 +160,7 @@ dependencies = [ [[package]] name = "atomic-canonical" -version = "0.18.2" +version = "0.18.3" dependencies = [ "atomic-identity", "blake3", @@ -174,7 +174,7 @@ dependencies = [ [[package]] name = "atomic-cli" -version = "0.18.2" +version = "0.18.3" dependencies = [ "anyhow", "atomic-agent", @@ -217,7 +217,7 @@ dependencies = [ [[package]] name = "atomic-config" -version = "0.18.2" +version = "0.18.3" dependencies = [ "anyhow", "dirs", @@ -231,7 +231,7 @@ dependencies = [ [[package]] name = "atomic-core" -version = "0.18.2" +version = "0.18.3" dependencies = [ "anyhow", "bitflags 2.13.0", @@ -258,7 +258,7 @@ dependencies = [ [[package]] name = "atomic-identity" -version = "0.18.2" +version = "0.18.3" dependencies = [ "anyhow", "atomic-config", @@ -280,7 +280,7 @@ dependencies = [ [[package]] name = "atomic-objects" -version = "0.18.2" +version = "0.18.3" dependencies = [ "blake3", "data-encoding", @@ -301,7 +301,7 @@ dependencies = [ [[package]] name = "atomic-remote" -version = "0.18.2" +version = "0.18.3" dependencies = [ "anyhow", "atomic-canonical", @@ -325,7 +325,7 @@ dependencies = [ [[package]] name = "atomic-repository" -version = "0.18.2" +version = "0.18.3" dependencies = [ "anyhow", "atomic-canonical", @@ -361,7 +361,7 @@ dependencies = [ [[package]] name = "atomic-semantic" -version = "0.18.2" +version = "0.18.3" dependencies = [ "serde", "serde_json", @@ -378,7 +378,7 @@ dependencies = [ [[package]] name = "atomic-teams" -version = "0.18.2" +version = "0.18.3" dependencies = [ "atomic-remote", "chrono", diff --git a/Cargo.toml b/Cargo.toml index bcc80d15..36780037 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -15,7 +15,7 @@ members = [ ] [workspace.package] -version = "0.18.2" +version = "0.18.3" edition = "2021" authors = ["Atomic Contributors"] license = "Apache-2.0" From 134f2714fb232221fb3388e36405d0a4bd5759b0 Mon Sep 17 00:00:00 2001 From: Sankalp Gilda Date: Tue, 22 Sep 2026 00:24:37 -0400 Subject: [PATCH 12/22] fix(canonical): make the declared depth cap the only depth cap Two depth limits were in force on the admission path and they disagreed by exactly one document. The admission options passed no explicit max_depth, so jcs_admit used its own default of 128 and accepted a document nested in 128 containers. The parse on the very next line, serde_json::from_slice, then applied a second limit of its own, also 128, but undeclared and refusing AT that depth rather than past it. A document sitting exactly on the cap was therefore admitted and then rejected one line later as one that "does not deserialize". The result is a validity split inside a single crate: it encodes a certificate it will not read back. encode_for_transport writes such a document into the Atomic-Delegation request header without complaint, and decode_from_transport refuses it on the way in, so the split is reachable by any caller that sends the header rather than only from a test. The error surfaced blames deserialization, pointing at the bytes instead of at the limit that actually fired, which is the wrong place to look. The cap is now declared once as jcs::MAX_DEPTH, passed to the admission explicitly, and serde_json's recursion limit is switched off for the parse. That is sound only because the admission has already walked the same bytes and refused anything nested deeper, so the parse never sees an input the cap did not already bound. The ordering is load-bearing and the code says so. What the crate admits is now exactly what it reads back, and the cap a caller relies on is the one that is written down. The three new vectors are wire bytes at 127, 128 and 129 containers, checked in as files so the boundary is exercised from both sides rather than only from past it. Each test asserts its file's length before reading it, so an editor that appends a newline cannot quietly change the document under test. --- atomic-canonical/Cargo.toml | 2 +- atomic-canonical/src/jcs.rs | 51 +++++-- atomic-canonical/tests/depth_boundary.rs | 129 ++++++++++++++++++ atomic-canonical/tests/ingest_boundary.rs | 13 +- .../tests/vectors/ingest/PROVENANCE.md | 25 ++++ .../tests/vectors/ingest/depth/127.json | 1 + .../tests/vectors/ingest/depth/128.json | 1 + .../tests/vectors/ingest/depth/129.json | 1 + 8 files changed, 205 insertions(+), 18 deletions(-) create mode 100644 atomic-canonical/tests/depth_boundary.rs create mode 100644 atomic-canonical/tests/vectors/ingest/depth/127.json create mode 100644 atomic-canonical/tests/vectors/ingest/depth/128.json create mode 100644 atomic-canonical/tests/vectors/ingest/depth/129.json diff --git a/atomic-canonical/Cargo.toml b/atomic-canonical/Cargo.toml index f5004d9f..d4bed667 100644 --- a/atomic-canonical/Cargo.toml +++ b/atomic-canonical/Cargo.toml @@ -10,7 +10,7 @@ rust-version.workspace = true [dependencies] serde = { workspace = true } -serde_json = { workspace = true } +serde_json = { workspace = true, features = ["unbounded_depth"] } jcs-admit = { workspace = true } blake3 = { workspace = true } bs58 = { workspace = true } diff --git a/atomic-canonical/src/jcs.rs b/atomic-canonical/src/jcs.rs index 2930e56e..716eb668 100644 --- a/atomic-canonical/src/jcs.rs +++ b/atomic-canonical/src/jcs.rs @@ -30,14 +30,23 @@ //! certificate as bytes and later hashes, signs or verifies it goes through this //! function, so the decision happens before the value exists. +use serde::Deserialize; use serde_json::Value; -use crate::error::Result; +use crate::error::{CanonicalError, Result}; + +/// The deepest nesting a document arriving as bytes is admitted at, counted in +/// open containers: the 129th nested array or object is refused as +/// [`jcs_admit::Error::TooDeep`] naming this value. It is the only depth bound +/// on the admission path, because [`admit_document`] parses with `serde_json`'s +/// own recursion limit switched off, so the cap declared here is the cap a +/// caller can rely on rather than one a parser default happens to apply. +pub const MAX_DEPTH: usize = 128; /// What a document arriving as bytes is admitted under. /// -/// RFC 8785 as written, plus the RFC 7493 I-JSON profile, plus integers only. -/// The profile is not decoration: +/// RFC 8785 as written, plus the RFC 7493 I-JSON profile, plus integers only, +/// under [`MAX_DEPTH`]. The profile is not decoration: /// /// - **Safe integers.** RFC 8785 section 3.2.2.3 defers number formatting to /// ECMAScript, which has one numeric type, so a conforming implementation @@ -47,18 +56,23 @@ use crate::error::Result; /// - **Integers only.** Nothing in the vocabulary needs a fractional number: /// the one numeric field a certificate carries is `maxChanges`, a count. Two /// implementations that never format a float can never disagree about one. -const ADMISSION: jcs_admit::Options = jcs_admit::Options::ijson().integers_only(true); +/// - **One depth cap.** [`MAX_DEPTH`] is set here rather than left to the +/// admission crate's default, because the parse that follows admission has no +/// depth bound of its own; whatever the constant says is the whole rule. +const ADMISSION: jcs_admit::Options = jcs_admit::Options::ijson() + .integers_only(true) + .max_depth(MAX_DEPTH); /// Admit a JSON document that arrived as bytes, and return the value it denotes. /// /// The refusals are the point. A repeated member name gives one document two /// readings: two parties take the same bytes for two different documents and a /// signature over either reading verifies, and by the time a `serde_json::Value` -/// exists the repeat is gone. Nesting past 128 containers is refused here rather -/// than walked, because the depth bound this crate's hashing and proof paths -/// rely on today is `serde_json`'s incidental parser default rather than one -/// either of them asserts. A number outside the profile above is refused rather -/// than rounded. +/// exists the repeat is gone. Nesting past [`MAX_DEPTH`] containers is refused +/// here rather than walked, and that is the only depth bound on the path: the +/// parse that follows runs with `serde_json`'s own recursion limit switched off, +/// so a document the admission accepts is a document this crate reads back. A +/// number outside the profile above is refused rather than rounded. /// /// The accepted document is parsed from the bytes it arrived in, not from the /// admission's canonical output, so nothing about an accepted document changes: @@ -73,9 +87,22 @@ const ADMISSION: jcs_admit::Options = jcs_admit::Options::ijson().integers_only( /// admission layer and `serde_json` rather than a statement about the input. pub fn admit_document(bytes: &[u8]) -> Result { jcs_admit::admit_with(bytes, &ADMISSION)?; - serde_json::from_slice(bytes).map_err(|e| { - crate::error::CanonicalError::Proof(format!("admitted document does not deserialize: {e}")) - }) + + // Sound only because the admission above has already walked these bytes and + // refused anything nested past MAX_DEPTH. serde_json's own recursion limit + // is switched off so the declared cap is the only cap: left on, it refuses + // at 128 while the admission admits 128, and a document this crate encodes + // is one it will not read back. This parse MUST stay ordered after + // `admit_with`. Parsed first, an input nested far enough would overflow + // the stack instead of returning an error. + let not_deserializable = |e: serde_json::Error| { + CanonicalError::Proof(format!("admitted document does not deserialize: {e}")) + }; + let mut de = serde_json::Deserializer::from_slice(bytes); + de.disable_recursion_limit(); + let value = Value::deserialize(&mut de).map_err(not_deserializable)?; + de.end().map_err(not_deserializable)?; + Ok(value) } /// Canonicalize a JSON value into its RFC-8785 string form. diff --git a/atomic-canonical/tests/depth_boundary.rs b/atomic-canonical/tests/depth_boundary.rs new file mode 100644 index 00000000..a6f9bbfa --- /dev/null +++ b/atomic-canonical/tests/depth_boundary.rs @@ -0,0 +1,129 @@ +//! The depth cap, tested on the cap rather than only past it. +//! +//! `jcs::admit_document` refuses nesting past `jcs::MAX_DEPTH`, and that is +//! meant to be the only depth bound on the path. The parse that follows the +//! admission used to carry a second one, `serde_json`'s recursion limit, which +//! refuses at 128 while the admission admits 128. A document exactly at the +//! declared cap was admitted, and then refused by the parse as one that "does +//! not deserialize"; `encode_for_transport` wrote it into the +//! `Atomic-Delegation` header and `decode_from_transport` would not read it +//! back. Every test here sits on the cap or one container either side of it, so +//! a second bound at any other value shows up as a failure rather than as a +//! document nobody happened to send. +//! +//! The three fixtures under `tests/vectors/ingest/depth/` are wire bytes: `[` +//! repeated *d* times, `null`, `]` repeated *d* times, no trailing newline. Each +//! test asserts the file's length before reading it, because an editor that +//! appends a newline changes the document under test. `PROVENANCE.md` next to +//! them says they were generated here rather than copied from the suite. + +use std::fs; +use std::path::{Path, PathBuf}; +use std::thread; + +use atomic_canonical::delegation; +use atomic_canonical::jcs::{self, MAX_DEPTH}; +use atomic_canonical::CanonicalError; +use jcs_admit::Error as Admission; +use serde_json::Value; + +/// The wire bytes of `depth/.json`, length-checked before anything reads them. +fn wire(d: usize) -> Vec { + let path: PathBuf = Path::new(env!("CARGO_MANIFEST_DIR")) + .join("tests/vectors/ingest/depth") + .join(format!("{d}.json")); + let bytes = fs::read(&path).unwrap_or_else(|e| panic!("read {path:?}: {e}")); + assert_eq!( + bytes.len(), + 2 * d + 4, + "{path:?} must be {d} open brackets, null, {d} close brackets and nothing else" + ); + bytes +} + +/// The value `depth/.json` denotes, built by a loop rather than by a parse. +fn nested(d: usize) -> Value { + (0..d).fold(Value::Null, |inner, _| Value::Array(vec![inner])) +} + +/// One under the cap: admitted, and it is the value the loop builds. +#[test] +fn one_under_the_cap_is_admitted() { + let bytes = wire(MAX_DEPTH - 1); + let admitted = jcs::admit_document(&bytes).expect("127 containers are under the cap"); + assert_eq!(admitted, nested(MAX_DEPTH - 1)); +} + +/// At the cap: admitted. This is the document the second bound refused. It is +/// the value the loop builds, and canonicalizing that value writes the wire +/// bytes back, so the fixture is its own canonical form. +#[test] +fn at_the_cap_is_admitted_and_canonicalizes_to_the_same_bytes() { + let bytes = wire(MAX_DEPTH); + let admitted = jcs::admit_document(&bytes).expect("128 containers are at the cap, not past it"); + assert_eq!(admitted, nested(MAX_DEPTH)); + assert_eq!( + jcs::canonicalize(&admitted).as_bytes(), + &bytes[..], + "the fixture is already canonical, so canonicalize must reproduce it byte for byte" + ); +} + +/// At the cap, through the header: what `encode_for_transport` writes, +/// `decode_from_transport` reads back, and encoding the result again gives the +/// same header. Before the declared cap became the only cap this was the +/// failing direction: encoded without complaint, refused on the way back in. +#[test] +fn at_the_cap_the_transport_round_trip_is_the_identity() { + let value = nested(MAX_DEPTH); + let encoded = delegation::encode_for_transport(&value); + let decoded = delegation::decode_from_transport(&encoded) + .expect("a document this crate encoded is one it reads back"); + assert_eq!(decoded, value); + assert_eq!(delegation::encode_for_transport(&decoded), encoded); +} + +/// One past the cap: refused as `TooDeep` naming `MAX_DEPTH`, at the byte where +/// the crossing container opens. Every bracket is one byte, so the 129th `[` +/// sits at byte 128. +#[test] +fn one_past_the_cap_is_refused_naming_the_declared_cap() { + let bytes = wire(MAX_DEPTH + 1); + match jcs::admit_document(&bytes) { + Err(CanonicalError::Admission(Admission::TooDeep { limit, offset })) => { + assert_eq!(limit, MAX_DEPTH); + assert_eq!(offset, MAX_DEPTH); + } + Err(other) => panic!("expected a depth refusal naming the declared cap, got {other}"), + Ok(_) => panic!("129 containers must be refused, the document was admitted"), + } +} + +/// The ordering guard. The admission bounds the depth before the parse runs +/// with `serde_json`'s recursion limit off, so a document 100,000 containers +/// deep is refused at the 129th and never reaches the unbounded parse. Run on +/// a thread with a 512 KiB stack: with the order reversed, that parse recurses +/// once per container, overflows the stack and aborts the whole test process, +/// which is a failure nobody can miss. +#[test] +fn far_past_the_cap_is_an_error_on_a_small_stack_and_never_an_abort() { + const DEPTH: usize = 100_000; + let mut bytes = vec![b'['; DEPTH]; + bytes.extend_from_slice(b"null"); + bytes.resize(2 * DEPTH + 4, b']'); + + let outcome = thread::Builder::new() + .name("depth-guard".into()) + .stack_size(512 * 1024) + .spawn(move || jcs::admit_document(&bytes).map(|_| ())) + .expect("spawn the guard thread") + .join() + .expect("the guard thread returns rather than panicking"); + match outcome { + Err(CanonicalError::Admission(Admission::TooDeep { limit, .. })) => { + assert_eq!(limit, MAX_DEPTH); + } + Err(other) => panic!("expected a depth refusal, got {other}"), + Ok(()) => panic!("100,000 containers must be refused, the document was admitted"), + } +} diff --git a/atomic-canonical/tests/ingest_boundary.rs b/atomic-canonical/tests/ingest_boundary.rs index adc5e307..a5c3cf1f 100644 --- a/atomic-canonical/tests/ingest_boundary.rs +++ b/atomic-canonical/tests/ingest_boundary.rs @@ -67,17 +67,20 @@ fn duplicate_member_is_refused() { assert_eq!(parsed["toolName"], "delete_repository"); } -/// `vd94ac70c9f0d84bf` (`aia-c-12`). One container past the 128 cap, refused -/// with a catchable error naming the cap rather than by recursing. The cap this -/// crate's hashing and proof paths rely on today is `serde_json`'s incidental -/// parser default, which is a bound neither of them asserts. +/// `vd94ac70c9f0d84bf` (`aia-c-12`). One container past the cap, refused with +/// a catchable error naming the cap rather than by recursing. The cap is +/// `jcs::MAX_DEPTH`, declared once and the only depth bound on the path; +/// `tests/depth_boundary.rs` sits on it from both sides. #[test] fn nesting_one_past_the_cap_is_refused() { let bytes = vector("statements/vd94ac70c9f0d84bf.json"); assert!( matches!( admission_error(&bytes), - Admission::TooDeep { limit: 128, .. } + Admission::TooDeep { + limit: jcs::MAX_DEPTH, + .. + } ), "vd94ac70c9f0d84bf must be refused at the depth cap" ); diff --git a/atomic-canonical/tests/vectors/ingest/PROVENANCE.md b/atomic-canonical/tests/vectors/ingest/PROVENANCE.md index ebf8304e..f78fe3f5 100644 --- a/atomic-canonical/tests/vectors/ingest/PROVENANCE.md +++ b/atomic-canonical/tests/vectors/ingest/PROVENANCE.md @@ -43,3 +43,28 @@ d94ac70c9f0d84bf1fc287376d1e2416785ce51df443a97a2d13765f5867883e statements/vd9 Two of the four are refused by RFC 8785 alone; the other two need the RFC 7493 profile `jcs::admit_document` applies. `tests/ingest_boundary.rs` says which is which, and asserts the variant rather than a message. + +## The depth cap from both sides, generated here + +Three documents that did not come from the conformance suite. They were +generated for this crate as the bytes `[` repeated *d* times, `null`, `]` +repeated *d* times, with no trailing newline, so each file is exactly 2*d* + 4 +bytes and `tests/depth_boundary.rs` asserts that length before it reads +anything else. `aia-c-12` above is one container past the cap on a real +statement; these sit on the cap itself, one either side, which is where a +second, undeclared depth bound shows up as a disagreement between the +admission and the parse that follows it. + +| file | depth | bytes | condition | expected | +| --- | --- | --- | --- | --- | +| `depth/127.json` | 127 | 258 | one under `jcs::MAX_DEPTH` | admitted | +| `depth/128.json` | 128 | 260 | at `jcs::MAX_DEPTH` | admitted; canonicalizes to the same bytes; identity through `encode_for_transport` and `decode_from_transport` | +| `depth/129.json` | 129 | 262 | one past `jcs::MAX_DEPTH` | `Error::TooDeep { limit: 128, offset: 128 }` | + +SHA-256 of each file as committed: + +``` +89fed8bdcec19b53a9cfeb6365642fcd7f84a12c0a8435224c1821f0183fa908 depth/127.json +9c2ec3e5c558bde9c95f21cd30b81503b5a76ab11f0a1bcbc9e55f45fa38bcf4 depth/128.json +aea6374322f648efe766b46b158da515b832005c4c2b2ba76de0f3f770bed699 depth/129.json +``` diff --git a/atomic-canonical/tests/vectors/ingest/depth/127.json b/atomic-canonical/tests/vectors/ingest/depth/127.json new file mode 100644 index 00000000..6d3f3917 --- /dev/null +++ b/atomic-canonical/tests/vectors/ingest/depth/127.json @@ -0,0 +1 @@ +[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[null]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]] \ No newline at end of file diff --git a/atomic-canonical/tests/vectors/ingest/depth/128.json b/atomic-canonical/tests/vectors/ingest/depth/128.json new file mode 100644 index 00000000..f9c044e3 --- /dev/null +++ b/atomic-canonical/tests/vectors/ingest/depth/128.json @@ -0,0 +1 @@ +[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[null]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]] \ No newline at end of file diff --git a/atomic-canonical/tests/vectors/ingest/depth/129.json b/atomic-canonical/tests/vectors/ingest/depth/129.json new file mode 100644 index 00000000..9f945842 --- /dev/null +++ b/atomic-canonical/tests/vectors/ingest/depth/129.json @@ -0,0 +1 @@ +[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[[null]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]]] \ No newline at end of file From d073ce675e6b408ea12bb87a2b83783f62e7a1c6 Mon Sep 17 00:00:00 2001 From: Bradley Hilton Date: Thu, 24 Sep 2026 10:59:23 -0700 Subject: [PATCH 13/22] feat(agent): add --from-skills to source skills inputs from a local checkout (#215) `atomic agent enable` always re-clones the skills-source package (atomic-skills) from Atomic storage on every install, even when --from is used. The manifest's [skills-source], [skills] and [agent-definition] inputs are only reachable through that remote sync, so offline installs of a local integration package still fail. Add --from-skills : point the skills inputs at a local atomic-skills checkout. No network access happens, and the manifest and agent definition are read from the given path. Without the flag, behavior is unchanged (remote sync, as before). --- atomic-cli/src/commands/agent/enable.rs | 43 +++++++++++++++++++++---- 1 file changed, 37 insertions(+), 6 deletions(-) diff --git a/atomic-cli/src/commands/agent/enable.rs b/atomic-cli/src/commands/agent/enable.rs index 9e898891..4accd42d 100644 --- a/atomic-cli/src/commands/agent/enable.rs +++ b/atomic-cli/src/commands/agent/enable.rs @@ -92,6 +92,17 @@ pub struct Enable { #[arg(long, value_name = "PATH")] from: Option, + /// Install the skills-source package (e.g. atomic-skills) from a local + /// checkout instead of syncing it from Atomic storage. + /// + /// The integration package's manifest can declare a + /// `[skills-source]` / `[skills]` / `[agent-definition]`. Without this + /// flag those inputs are always re-cloned from Atomic storage on every + /// install, even when `--from` is given. Point this at a local + /// atomic-skills checkout to install fully offline. + #[arg(long, value_name = "PATH")] + from_skills: Option, + /// Also install AGENTS.md into the repository root so the Atomic /// workflow is always-on without picking a bundled agent. /// @@ -113,6 +124,7 @@ impl Enable { global: false, hooks: None, from: None, + from_skills: None, agents_md: false, } } @@ -422,7 +434,7 @@ impl Command for Enable { println!(" Turns are still recorded by the built-in hooks, but these agents"); println!(" are missing their Atomic skills and system prompt."); println!(" Retry with 'atomic agent enable --force' once the package is reachable,"); - println!(" or install from a local checkout with '--from '."); + println!(" or install from a local checkout with '--from ' (and '--from-skills ' for the skills package)."); return Err(partial_install_error(°raded)); } @@ -437,7 +449,7 @@ impl Command for Enable { /// is **always** re-cloned on every install. This ensures new skills added /// to atomic-skills are picked up immediately without --force. If the /// re-clone fails (network), the error is returned — the caller can retry -/// or use --from to install from a local checkout. +/// or use --from-skills to source the skills package from a local checkout. fn sync_skills_cache(_force: bool) -> CliResult { const SKILLS_AGENT: &str = "atomic-skills"; @@ -512,9 +524,11 @@ impl Enable { /// package's atomic-integration.toml. /// /// When the manifest declares `[skills-source]`, the shared atomic-skills - /// cache is synced (or reused) and passed as `skills_cache_dir`. When the - /// user opts in via `--agents-md` or the prompt, the repo root is passed - /// so `[[repo-file]]` entries land in the repo. + /// cache is synced (or reused) and passed as `skills_cache_dir`. With + /// `--from-skills` the skills inputs come from a local checkout and no + /// network access happens at all. When the user opts in via + /// `--agents-md` or the prompt, the repo root is passed so + /// `[[repo-file]]` entries land in the repo. fn install_integration( &self, agent_name: &str, @@ -538,7 +552,20 @@ impl Enable { || manifest.agent_definition.is_some() || !manifest.skills.is_empty() { - Some(sync_skills_cache(self.force)?) + if let Some(ref from_skills) = self.from_skills { + if !std::fs::exists(from_skills).unwrap_or(false) { + return Err(crate::error::CliError::InvalidArgument { + message: format!( + "--from-skills: no checkout at {}; point it at a local \ + atomic-skills checkout (or omit it to sync from storage)", + from_skills.display() + ), + }); + } + Some(from_skills.clone()) + } else { + Some(sync_skills_cache(self.force)?) + } } else { None }; @@ -836,6 +863,7 @@ mod tests { global: false, hooks: None, from: None, + from_skills: None, agents_md: false, }; } @@ -849,6 +877,7 @@ mod tests { global: false, hooks: None, from: None, + from_skills: None, agents_md: false, }; assert!(cmd.force); @@ -864,6 +893,7 @@ mod tests { global: false, hooks: None, from: None, + from_skills: None, agents_md: false, }; assert!(cmd.all); @@ -879,6 +909,7 @@ mod tests { global: true, hooks: None, from: None, + from_skills: None, agents_md: false, }; assert!(cmd.global); From 22de671ba3185b431b9eae75911614adb0ca6766 Mon Sep 17 00:00:00 2001 From: Bradley Hilton Date: Fri, 25 Sep 2026 13:37:46 -0700 Subject: [PATCH 14/22] feat(diff): add --json, and point a clean working copy at -c (#218) `atomic diff` conflates two questions: working copy vs. recorded state (no -c), and the state before vs. after a change (-c). It also had no machine-readable mode, making it the only core VCS command without one. - Add `--json`, emitting a versioned document (per-file status and paths, hunks with typed lines, insertion/deletion rollups). It takes precedence over --stat/--name-only/--name-status so consumers get one parseable document instead of formatted output to re-parse. A `-c` diff carries the change header under `change`; a working-copy diff carries `view`. An empty diff still emits valid JSON. - When the working copy is clean, name real copy-pasteable `-c` commands for the most recent changes on the current view. A clean working copy was a dead end for anyone expecting to see a change. Uses include_inherited so a freshly forked draft does not claim it has no recorded changes. --- atomic-cli/src/commands/diff/command.rs | 39 ++- atomic-cli/src/commands/diff/helpers.rs | 134 ++++++++- atomic-cli/src/commands/diff/json.rs | 344 ++++++++++++++++++++++++ atomic-cli/src/commands/diff/mod.rs | 72 ++++- 4 files changed, 561 insertions(+), 28 deletions(-) create mode 100644 atomic-cli/src/commands/diff/json.rs diff --git a/atomic-cli/src/commands/diff/command.rs b/atomic-cli/src/commands/diff/command.rs index 243a9e89..5a67eed7 100644 --- a/atomic-cli/src/commands/diff/command.rs +++ b/atomic-cli/src/commands/diff/command.rs @@ -7,11 +7,13 @@ use super::*; // Diff Command -/// Show changes between working copy and repository. +/// Show changes. /// -/// The `diff` command compares the current state of files in the working -/// copy against their recorded state in the repository, displaying the -/// differences in a human-readable format. +/// With `-c `, displays what a specific recorded change did, by +/// comparing the view state before it against the state after it. +/// +/// Without `-c`, compares the current working copy against the recorded +/// state of the current view. /// /// # Output Formats /// @@ -19,6 +21,8 @@ use super::*; /// - **Stat**: Summary showing files and line counts /// - **Name-only**: Just file paths /// - **Name-status**: File paths with status indicators +/// - **JSON** (`--json`): Versioned machine-readable document; takes +/// precedence over the text formats /// /// # Algorithms /// @@ -32,6 +36,8 @@ pub struct Diff { pub files: Vec, /// Compare against a specific change hash or prefix. + /// + /// Omit this to diff the working copy against the current view instead. #[arg(short = 'c', long = "change", add = ArgValueCompleter::new(complete_change_hashes))] pub change: Option, @@ -81,6 +87,10 @@ pub struct Diff { /// that the line changed. Especially useful for code reviews. #[arg(long)] pub word_diff: bool, + + /// Emit a versioned JSON document instead of a rendered diff. + #[arg(long)] + pub json: bool, } impl Diff { @@ -100,6 +110,7 @@ impl Diff { cached: false, view: None, word_diff: false, + json: false, } } @@ -167,6 +178,12 @@ impl Diff { self } + /// Builder: set the JSON output flag. + pub fn with_json(mut self, json: bool) -> Self { + self.json = json; + self + } + /// Get the output format based on command flags. pub fn get_format(&self) -> DiffFormat { if self.name_only { @@ -242,6 +259,11 @@ impl Command for Diff { return self.show_change_diff(&repo, change_ref, &config); } + let view = self + .view + .clone() + .unwrap_or_else(|| repo.current_view().to_string()); + // Get status to find modified files let status_options = StatusOptions::default(); let status = repo @@ -286,7 +308,7 @@ impl Command for Diff { // Check if there are any changes if files_to_diff.is_empty() { - self.print_no_changes(); + self.print_no_pending_changes(&repo, &view); return Ok(()); } @@ -444,11 +466,6 @@ impl Command for Diff { } // Print in the appropriate format - match config.format { - DiffFormat::Unified => self.print_unified(&file_diffs, &config), - DiffFormat::Stat => self.print_stat(&stats, &config), - DiffFormat::NameOnly => self.print_name_only(&file_diffs), - DiffFormat::NameStatus => self.print_name_status(&file_diffs, &config), - } + self.render(&file_diffs, &stats, &config, None, Some(view.as_str())) } } diff --git a/atomic-cli/src/commands/diff/helpers.rs b/atomic-cli/src/commands/diff/helpers.rs index 6abaa25e..ad7a1b95 100644 --- a/atomic-cli/src/commands/diff/helpers.rs +++ b/atomic-cli/src/commands/diff/helpers.rs @@ -66,11 +66,119 @@ impl NumberedLine { } impl Diff { + /// Render a computed diff in the requested format. + /// + /// `--json` wins over every text format selector, so a consumer gets a + /// single parseable document rather than a rendered diff it has to + /// re-parse. `change` is `Some` for `atomic diff -c `; `view` is + /// populated only for working-copy diffs. + pub(super) fn render( + &self, + file_diffs: &[FileDiff], + stats: &DiffStats, + config: &DiffOutputConfig, + change: Option<(&Change, &Hash)>, + view: Option<&str>, + ) -> CliResult<()> { + if self.json { + return json::print_json(&json::JsonDiff::new(file_diffs, stats, change, view)); + } + + match config.format { + DiffFormat::Unified => self.print_unified(file_diffs, config), + DiffFormat::Stat => self.print_stat(stats, config), + DiffFormat::NameOnly => self.print_name_only(file_diffs), + DiffFormat::NameStatus => self.print_name_status(file_diffs, config), + } + } + /// Print a message when there are no changes. pub(super) fn print_no_changes(&self) { + if self.json { + let _ = json::print_json(&json::JsonDiff::empty(None)); + return; + } print_info("No changes detected"); } + /// Explain that the working copy matches the view, and point at the + /// change records that *are* inspectable. + /// + /// `atomic diff` with no `-c` compares disk against the view's recorded + /// state. A clean working copy is a perfectly valid answer, but it is a + /// dead end for someone who typed `atomic diff` expecting to see a + /// change. Naming a real, copy-pasteable `-c` command for the most + /// recent changes on this view turns the dead end into a next step. + pub(super) fn print_no_pending_changes(&self, repo: &Repository, view: &str) { + if self.json { + let _ = json::print_json(&json::JsonDiff::empty(Some(view))); + return; + } + + print_info("No changes detected"); + println!(); + + let recent = self.recent_view_changes(repo, view, RECENT_CHANGE_SUGGESTIONS); + + if recent.is_empty() { + print_hint(&format!( + "Working copy matches view `{view}`, which has no recorded changes yet." + )); + print_hint("Record the working copy with `atomic record -m \"\"`."); + return; + } + + print_hint(&format!( + "Working copy matches view `{view}`. To inspect a recorded change:" + )); + for (short_hash, message) in &recent { + print_hint(&format!(" atomic diff -c {short_hash} {message}")); + } + print_hint(" atomic log — list the changes on this view"); + } + + /// Collect up to `limit` recent changes on `view`, newest first, as + /// `(short_hash, message)` pairs for the no-pending-changes hint. + /// + /// Best-effort: any failure to read history yields an empty list, which + /// degrades the hint rather than the diff. + fn recent_view_changes( + &self, + repo: &Repository, + view: &str, + limit: usize, + ) -> Vec<(String, String)> { + use atomic_repository::history::HistoryOptions; + + let options = HistoryOptions::new() + .view(view.to_string()) + .limit(limit) + .load_headers(true) + // A draft view's own log is usually empty right after it forks, + // but it still inherits everything from its ancestors — and those + // changes are diffable here. Without this the hint would claim + // the view has no recorded changes when it plainly does. + .include_inherited(true); + + let Ok(entries) = repo.reverse_log(options) else { + return Vec::new(); + }; + + entries + .into_iter() + .map(|entry| { + let base32 = entry.hash.to_base32(); + let short_hash = base32[..DEFAULT_HASH_LENGTH.min(base32.len())].to_string(); + let message = entry + .header + .map(|h| h.message) + .filter(|m| !m.trim().is_empty()) + .unwrap_or_else(|| "(no message)".to_string()); + (short_hash, message) + }) + .collect() + } + /// Check whether a repository-relative path passes the positional /// file filter (`atomic diff --change ...`). /// @@ -358,16 +466,19 @@ impl Diff { return Ok(()); } - if config.format == DiffFormat::Unified { + // The human header duplicates what `--json` already carries under + // `change`, so skip it there. + if !self.json && config.format == DiffFormat::Unified { self.print_change_header(change, change_hash, config); } - match config.format { - DiffFormat::Unified => self.print_unified(&file_diffs, config), - DiffFormat::Stat => self.print_stat(&stats, config), - DiffFormat::NameOnly => self.print_name_only(&file_diffs), - DiffFormat::NameStatus => self.print_name_status(&file_diffs, config), - } + self.render( + &file_diffs, + &stats, + config, + Some((change, change_hash)), + None, + ) } /// Reconstruct a line's text content from its leaf operations. @@ -756,17 +867,12 @@ impl Diff { } // Print change header information - if config.format == DiffFormat::Unified { + if !self.json && config.format == DiffFormat::Unified { self.print_change_header(change, hash, config); } // Print in the appropriate format - match config.format { - DiffFormat::Unified => self.print_unified(&file_diffs, config), - DiffFormat::Stat => self.print_stat(&stats, config), - DiffFormat::NameOnly => self.print_name_only(&file_diffs), - DiffFormat::NameStatus => self.print_name_status(&file_diffs, config), - } + self.render(&file_diffs, &stats, config, Some((change, hash)), None) } /// Print header information for a change diff. diff --git a/atomic-cli/src/commands/diff/json.rs b/atomic-cli/src/commands/diff/json.rs new file mode 100644 index 00000000..b27551d9 --- /dev/null +++ b/atomic-cli/src/commands/diff/json.rs @@ -0,0 +1,344 @@ +//! Machine-readable JSON output for `atomic diff --json`. +//! +//! The document is a versioned projection of the same [`FileDiff`] data the +//! text renderers consume, so a consumer can reconstruct a unified diff or +//! compute its own rollups without re-parsing formatted output. + +use super::*; +use serde::Serialize; + +/// Version of the `atomic diff --json` document. +const DIFF_JSON_SCHEMA_VERSION: u32 = 1; + +/// The root document. +#[derive(Debug, Serialize)] +pub(super) struct JsonDiff { + schema_version: u32, + + /// The view the diff was taken against, when the diff is working-copy + /// scoped. `None` for a `-c` change diff, which is not view-scoped. + view: Option, + + /// The change record this diff describes, when `-c` was given. + change: Option, + + files: Vec, + stats: JsonDiffStats, +} + +impl JsonDiff { + /// Build the document. + /// + /// `change` is `Some` for `atomic diff -c ` and `None` for a + /// working-copy diff; `view` is populated only for the latter. + pub(super) fn new( + file_diffs: &[FileDiff], + stats: &DiffStats, + change: Option<(&Change, &Hash)>, + view: Option<&str>, + ) -> Self { + Self { + schema_version: DIFF_JSON_SCHEMA_VERSION, + view: view.map(str::to_string), + change: change.map(|(c, h)| JsonDiffChange::new(c, h)), + files: file_diffs.iter().map(JsonDiffFile::from).collect(), + stats: JsonDiffStats::from(stats), + } + } + + /// The all-zero document, used when there is nothing to diff so that + /// `--json` still emits valid parseable JSON. + pub(super) fn empty(view: Option<&str>) -> Self { + Self { + schema_version: DIFF_JSON_SCHEMA_VERSION, + view: view.map(str::to_string), + change: None, + files: Vec::new(), + stats: JsonDiffStats::default(), + } + } +} + +/// Identity and message of the change record under inspection. +#[derive(Debug, Serialize)] +struct JsonDiffChange { + /// Full base32 change hash (not truncated — this is a machine surface). + hash: String, + + /// Short hash as displayed by the text renderers. + short_hash: String, + + message: String, + description: Option, + authors: Vec, + date: String, +} + +impl JsonDiffChange { + fn new(change: &Change, hash: &Hash) -> Self { + let header = &change.hashed.header; + let base32 = hash.to_base32(); + Self { + short_hash: base32[..DEFAULT_HASH_LENGTH.min(base32.len())].to_string(), + hash: base32, + message: header.message.clone(), + description: header.description.clone(), + authors: header.authors.iter().map(JsonDiffAuthor::from).collect(), + date: header.timestamp.format("%Y-%m-%dT%H:%M:%SZ").to_string(), + } + } +} + +#[derive(Debug, Serialize)] +struct JsonDiffAuthor { + name: String, + email: Option, +} + +impl From<&atomic_core::change::Author> for JsonDiffAuthor { + fn from(author: &atomic_core::change::Author) -> Self { + Self { + name: author.name.clone(), + email: author.email.clone(), + } + } +} + +/// One file's changes. +#[derive(Debug, Serialize)] +struct JsonDiffFile { + /// Repository-relative path, taken from whichever side exists. + path: String, + + /// `/dev/null` when the file did not exist before. + old_path: String, + + /// `/dev/null` when the file no longer exists after. + new_path: String, + + /// `added`, `deleted`, `modified`, `renamed`, `untracked`, … + status: &'static str, + + /// The conventional single-character status (`A`, `D`, `M`, …). + code: char, + + binary: bool, + insertions: usize, + deletions: usize, + hunks: Vec, +} + +impl From<&FileDiff> for JsonDiffFile { + fn from(diff: &FileDiff) -> Self { + Self { + path: diff.display_path().to_string(), + old_path: diff.old_path.clone(), + new_path: diff.new_path.clone(), + status: diff.status.description(), + code: diff.status.status_char(), + binary: diff.is_binary, + insertions: diff.stats.insertions, + deletions: diff.stats.deletions, + hunks: diff.hunks.iter().map(JsonDiffHunk::from).collect(), + } + } +} + +/// A contiguous region of change, matching unified-diff `@@` semantics. +#[derive(Debug, Serialize)] +struct JsonDiffHunk { + old_start: usize, + old_count: usize, + new_start: usize, + new_count: usize, + lines: Vec, +} + +impl From<&DiffHunk> for JsonDiffHunk { + fn from(hunk: &DiffHunk) -> Self { + Self { + old_start: hunk.old_start, + old_count: hunk.old_count, + new_start: hunk.new_start, + new_count: hunk.new_count, + lines: hunk.lines.iter().map(JsonDiffLine::from).collect(), + } + } +} + +#[derive(Debug, Serialize)] +struct JsonDiffLine { + /// `context`, `added`, or `removed`. + status: &'static str, + + /// Line content without its trailing newline. + content: String, + + /// 1-based line number in the pre-image; `None` for additions. + old_line: Option, + + /// 1-based line number in the post-image; `None` for removals. + new_line: Option, +} + +impl From<&HunkLine> for JsonDiffLine { + fn from(line: &HunkLine) -> Self { + Self { + status: line_status_name(line.status), + content: line.content.clone(), + old_line: line.old_line_num, + new_line: line.new_line_num, + } + } +} + +/// Whole-diff rollups. +#[derive(Debug, Serialize, Default)] +struct JsonDiffStats { + files: usize, + insertions: usize, + deletions: usize, +} + +impl From<&DiffStats> for JsonDiffStats { + fn from(stats: &DiffStats) -> Self { + Self { + files: stats.file_count(), + insertions: stats.total_insertions(), + deletions: stats.total_deletions(), + } + } +} + +fn line_status_name(status: LineStatus) -> &'static str { + match status { + LineStatus::Unchanged => "context", + LineStatus::Added => "added", + LineStatus::Removed => "removed", + } +} + +/// Print the document to stdout. +pub(super) fn print_json(document: &JsonDiff) -> CliResult<()> { + let rendered = + serde_json::to_string_pretty(document).map_err(|e| CliError::Internal(e.into()))?; + println!("{rendered}"); + Ok(()) +} + +#[cfg(test)] +mod tests { + use super::*; + + fn sample_file_diff() -> FileDiff { + let mut diff = FileDiff::modified("src/main.rs"); + let mut hunk = DiffHunk::new(1, 2, 1, 2); + hunk.add_line(HunkLine::context("fn main() {", 1, 1)); + hunk.add_line(HunkLine::removed(" println!(\"hi\");", 2)); + hunk.add_line(HunkLine::added(" println!(\"hi, world\");", 2)); + diff.add_hunk(hunk); + diff.stats = FileDiffStats::modified("src/main.rs", 1, 1); + diff + } + + fn sample_stats(files: &[FileDiff]) -> DiffStats { + let mut stats = DiffStats::new(); + for f in files { + stats.add_file(f.stats.clone()); + } + stats + } + + #[test] + fn empty_document_is_valid_json_with_zero_stats() { + let doc = JsonDiff::empty(Some("main")); + let value: serde_json::Value = + serde_json::to_value(&doc).expect("empty document serializes"); + + assert_eq!(value["schema_version"], 1); + assert_eq!(value["view"], "main"); + assert!(value["change"].is_null()); + assert_eq!(value["files"].as_array().unwrap().len(), 0); + assert_eq!(value["stats"]["files"], 0); + assert_eq!(value["stats"]["insertions"], 0); + assert_eq!(value["stats"]["deletions"], 0); + } + + #[test] + fn file_diffs_project_to_per_file_hunks_and_lines() { + let diffs = vec![sample_file_diff()]; + let doc = JsonDiff::new(&diffs, &sample_stats(&diffs), None, Some("main")); + let value: serde_json::Value = serde_json::to_value(&doc).expect("serializes"); + + let file = &value["files"][0]; + assert_eq!(file["path"], "src/main.rs"); + assert_eq!(file["status"], "modified"); + assert_eq!(file["code"], "M"); + assert_eq!(file["insertions"], 1); + assert_eq!(file["deletions"], 1); + assert_eq!(file["binary"], false); + + let hunk = &file["hunks"][0]; + assert_eq!(hunk["old_start"], 1); + assert_eq!(hunk["new_count"], 2); + + let lines = hunk["lines"].as_array().unwrap(); + assert_eq!(lines.len(), 3); + assert_eq!(lines[0]["status"], "context"); + assert_eq!(lines[0]["old_line"], 1); + assert_eq!(lines[0]["new_line"], 1); + + // A removal has no post-image line number, and vice versa. + assert_eq!(lines[1]["status"], "removed"); + assert_eq!(lines[1]["old_line"], 2); + assert!(lines[1]["new_line"].is_null()); + + assert_eq!(lines[2]["status"], "added"); + assert_eq!(lines[2]["new_line"], 2); + assert!(lines[2]["old_line"].is_null()); + } + + #[test] + fn aggregate_stats_roll_up_across_files() { + let diffs = vec![sample_file_diff(), sample_file_diff()]; + let doc = JsonDiff::new(&diffs, &sample_stats(&diffs), None, None); + let value: serde_json::Value = serde_json::to_value(&doc).expect("serializes"); + + assert_eq!(value["stats"]["files"], 2); + assert_eq!(value["stats"]["insertions"], 2); + assert_eq!(value["stats"]["deletions"], 2); + assert!(value["view"].is_null()); + } + + #[test] + fn added_file_reports_dev_null_as_old_path() { + let diff = FileDiff::added("src/new.rs"); + let mut added = diff.clone(); + added.stats = FileDiffStats::added("src/new.rs", 3); + let diffs = vec![added]; + let doc = JsonDiff::new(&diffs, &sample_stats(&diffs), None, None); + let value: serde_json::Value = serde_json::to_value(&doc).expect("serializes"); + + let file = &value["files"][0]; + assert_eq!(file["old_path"], "/dev/null"); + assert_eq!(file["new_path"], "src/new.rs"); + assert_eq!(file["path"], "src/new.rs"); + assert_eq!(file["status"], "added"); + assert_eq!(file["code"], "A"); + } + + #[test] + fn deleted_file_reports_dev_null_as_new_path() { + let mut diff = FileDiff::deleted("src/old.rs"); + diff.stats = FileDiffStats::deleted("src/old.rs", 2); + let diffs = vec![diff]; + let doc = JsonDiff::new(&diffs, &sample_stats(&diffs), None, None); + let value: serde_json::Value = serde_json::to_value(&doc).expect("serializes"); + + let file = &value["files"][0]; + assert_eq!(file["old_path"], "src/old.rs"); + assert_eq!(file["new_path"], "/dev/null"); + assert_eq!(file["status"], "deleted"); + assert_eq!(file["code"], "D"); + } +} diff --git a/atomic-cli/src/commands/diff/mod.rs b/atomic-cli/src/commands/diff/mod.rs index eecb4f99..724c9769 100644 --- a/atomic-cli/src/commands/diff/mod.rs +++ b/atomic-cli/src/commands/diff/mod.rs @@ -34,18 +34,75 @@ //! [FILES]... Specific files to diff (default: all modified files) //! //! Options: -//! -c, --change Compare against a specific change +//! -c, --change Show a specific recorded change //! --algorithm Diff algorithm (myers, patience) [default: myers] //! --context Number of context lines [default: 3] //! --stat Show diffstat summary only //! --no-color Disable colored output //! --word-diff Enable token-level diff highlighting -//! --cached Show staged changes (not yet implemented) +//! --json Emit a versioned JSON document //! --name-only Show only names of changed files //! --name-status Show names and status of changed files //! -h, --help Print help information //! ``` //! +//! # Working Copy vs. Recorded Change +//! +//! `atomic diff` compares two different things depending on whether `-c` is +//! given, and the distinction is the whole point of the command: +//! +//! | Invocation | Left side | Right side | Answers | +//! |---|---|---|---| +//! | `atomic diff` | the view's **recorded** state | your **working copy** on disk | "what have I changed but not recorded yet?" | +//! | `atomic diff -c ` | the state **before** that change | the state **after** it | "what did change record `` do?" | +//! +//! When the working copy is clean, `atomic diff` says so and names real, +//! copy-pasteable `-c` commands for the most recent changes on the current +//! view, so a clean working copy is a next step rather than a dead end. +//! +//! # JSON Output +//! +//! `--json` emits a versioned document and takes precedence over every text +//! format selector (`--stat`, `--name-only`, `--name-status`): +//! +//! ```text +//! $ atomic diff -c 5XGIB2VGQRAF --json +//! { +//! "schema_version": 1, +//! "view": null, +//! "change": { +//! "hash": "5XGIB2VGQRAFC334ZU6LHH23X6FBST2S2I7GCC7SEFVO62KZ634A", +//! "short_hash": "5XGIB2VGQRAF", +//! "message": "Greet the world", +//! "authors": [{ "name": "bradley", "email": "bradley.hilton@atomic.dev" }], +//! "date": "2026-09-25T18:23:15Z" +//! }, +//! "files": [{ +//! "path": "main.rs", +//! "old_path": "main.rs", +//! "new_path": "main.rs", +//! "status": "modified", +//! "code": "M", +//! "insertions": 2, +//! "deletions": 1, +//! "hunks": [{ +//! "old_start": 1, "old_count": 4, +//! "new_start": 1, "new_count": 5, +//! "lines": [ +//! { "status": "removed", "content": " println!(\"hi\");", "old_line": 2, "new_line": null }, +//! { "status": "added", "content": " println!(\"hi, world\");", "old_line": null, "new_line": 2 } +//! ] +//! }] +//! }], +//! "stats": { "files": 1, "insertions": 2, "deletions": 1 } +//! } +//! ``` +//! +//! `change` is `null` and `view` is set for working-copy diffs; `view` is +//! `null` and `change` is set for `-c` diffs. `old_path`/`new_path` are +//! `/dev/null` on the absent side of an add or delete. A diff with nothing to +//! show still emits valid JSON with an empty `files` array. +//! //! # Output Formats //! //! ## Default (Unified Diff) @@ -104,6 +161,11 @@ //! ... //! ``` //! +//! Inspect a recorded change, as JSON: +//! ```text +//! $ atomic diff -c 5XGIB2VGQRAF --json +//! ``` +//! //! Show changes for specific file: //! ```text //! $ atomic diff src/main.rs @@ -167,15 +229,19 @@ use atomic_repository::Repository; use crate::commands::{find_repository_root, Command, DEFAULT_HASH_LENGTH}; use crate::error::{CliError, CliResult}; use crate::output::{ - added, deleted, emphasis, hash, info, modified, path as style_path, print_info, + added, deleted, emphasis, hash, info, modified, path as style_path, print_hint, print_info, }; mod command; mod format; mod helpers; +mod json; mod output; mod types; +/// How many recent changes the no-pending-changes hint offers to inspect. +const RECENT_CHANGE_SUGGESTIONS: usize = 3; + pub use command::*; pub(crate) use helpers::change_file_diffs; pub(crate) use output::{build_hunks_from_diff, format_stat_graph}; From ba802e04220afaadda2ca67971217998127d7320 Mon Sep 17 00:00:00 2001 From: Aaron Ogle Date: Fri, 25 Sep 2026 15:49:25 -0500 Subject: [PATCH 15/22] feat(agent): select a delegated identity for hook recording (#219) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit * feat(agent): select a delegated identity for hook recording `atomic agent identity set ` makes every hooked agent on this machine record under a delegated identity's own key instead of the plus-tag of the default identity. The selection is global — one identity per machine, deliberately not per repository — written to a new top-level `agent_identity` in ~/.atomic/config.toml. Hooks resolve the identity on every invocation: ATOMIC_AGENT_IDENTITY env var > global agent_identity setting > active server profile's agent_identity binding > plus-tag fallback The server-profile link closes a promise that was already written down: `bind_agent_identity` documents the binding as "so hooks use it by default", but only push auth ever read it — recording never did. `set` validates eagerly (the identity must resolve and be an agent/delegated type; human identities are refused; a missing delegation certificate warns) while the record path degrades softly on a bad name — recording a turn must never fail over identity selection. `show` and `agent status` (human + JSON) report the effective identity and its source. With nothing configured anywhere, recorded turns are indistinguishable from before, and `atomic record` is untouched. atomic-agent never reads config: the resolved name crosses as plain data through TurnRecordOptions into build_agent_author and active_delegation_urn, so change headers carry the agent's public key and envelopes name the active delegation certificate. * chore(agent): fmt and rustdoc fixes for identity selection cargo fmt across the touched crates, and the `key()` doc comment linked to `Self::fmt` — a trait impl method rustdoc cannot resolve — which failed the Documentation job under -Dwarnings. * fix(cli): serialize database-owner integration tests Every test in database_owner_integration_test spawns real atomic subprocesses that hold the redb lock and burn CPU. On 2-core Windows CI runners, the default test-threads parallelism lets the process-heavy tests (the eight-process session-start hammer, the failpoint owners) starve whichever sibling tests overlap them past their database-wait budgets. The failure signature is exactly that: different tests fail on different runs — concurrent_session_starts…, crashing_second_checkpoint…, owner_death_after_checkpoint_prepare… in this PR's runs; concurrent_stops_publish… on another PR the same day — all in this file, all contention-shaped (one with an explicit 'Database already open. Cannot acquire lock'), while macOS and ubuntu pass the same suite. #[serial] trades a few minutes of wall time for runs that only fail when something is actually broken. Locally the serialized suite passes in 53s (macOS); CI is the only place the starvation reproduced. --- atomic-agent/src/identity.rs | 43 +- atomic-agent/src/record/mod.rs | 4 +- atomic-agent/src/record/options.rs | 11 + atomic-agent/src/record/provenance.rs | 4 +- atomic-agent/src/record/tests.rs | 4 + atomic-agent/src/turn/orchestrator/mod.rs | 23 + .../src/turn/orchestrator/session_end.rs | 1 + atomic-agent/src/turn/orchestrator/turn.rs | 2 + .../tests/record_view_mismatch_regression.rs | 1 + atomic-cli/src/commands/agent/hooks.rs | 18 + atomic-cli/src/commands/agent/identity.rs | 454 ++++++++++++++++++ atomic-cli/src/commands/agent/mod.rs | 26 + atomic-cli/src/commands/agent/status.rs | 39 +- .../tests/database_owner_integration_test.rs | 35 ++ atomic-config/src/lib.rs | 93 ++++ 15 files changed, 753 insertions(+), 5 deletions(-) create mode 100644 atomic-cli/src/commands/agent/identity.rs diff --git a/atomic-agent/src/identity.rs b/atomic-agent/src/identity.rs index 0367c60c..34831dd0 100644 --- a/atomic-agent/src/identity.rs +++ b/atomic-agent/src/identity.rs @@ -494,6 +494,11 @@ fn extract_toml_string_value(line: &str) -> Option { /// * `agent_name` — Agent registry key (e.g., "claude-code") /// * `agent_display_name` — Human-readable name (e.g., "Claude Code") /// * `session_id` — Session identifier for the `+tag` suffix +/// * `agent_identity` — Name of a delegated agent identity to sign as, +/// resolved by the CLI hook handler (env var > global setting > active +/// server profile). `None` falls back to `ATOMIC_AGENT_IDENTITY` and then +/// to the plus-tag path, so an environment with nothing configured +/// behaves exactly as before agent identities existed. /// /// # Returns /// @@ -502,13 +507,18 @@ fn extract_toml_string_value(line: &str) -> Option { /// /// - With identity: `claude+60f5 ` (with public key ref) /// - Without: `Claude Code` (no email) -pub fn build_agent_author(agent_name: &str, agent_display_name: &str, session_id: &str) -> Author { +pub fn build_agent_author( + agent_name: &str, + agent_display_name: &str, + session_id: &str, + agent_identity: Option<&str>, +) -> Author { let options = AgentAuthorOptions { agent_name, agent_display_name, session_id, identity_dir: None, - agent_identity: None, + agent_identity: agent_identity.map(str::to_string), }; resolve_agent_author(&options) } @@ -943,6 +953,7 @@ identity_type = "user" "claude-code", "Claude Code", "60f5cbd2-aa23-40ee-9085-4375dd186ce7", + None, ); // Can't guarantee identity store exists in test env, so just verify @@ -950,6 +961,34 @@ identity_type = "user" assert!(!author.name.is_empty()); } + /// A name that does not resolve must degrade to a usable author, not an + /// error — recording a turn must never fail over identity selection. + #[test] + fn build_agent_author_never_fails_on_an_unresolvable_name() { + let author = build_agent_author("open-code", "OpenCode", "sess1234", Some("no-such-agent")); + assert!(!author.name.is_empty()); + } + + /// The delegated name must reach the resolver: `build_agent_author` is a + /// thin wrapper over `resolve_agent_author`, so the same inputs — with + /// and without a delegated name — must produce identical authors. + #[test] + fn build_agent_author_threads_the_delegated_name() { + for name in [None, Some("fred+opencode")] { + let via_wrapper = build_agent_author("open-code", "OpenCode", "sess1234", name); + let via_resolver = resolve_agent_author(&AgentAuthorOptions { + agent_name: "open-code", + agent_display_name: "OpenCode", + session_id: "sess1234", + identity_dir: None, + agent_identity: name.map(str::to_string), + }); + assert_eq!(via_wrapper.name, via_resolver.name); + assert_eq!(via_wrapper.email, via_resolver.email); + assert_eq!(via_wrapper.identity, via_resolver.identity); + } + } + // extract_toml_string_value #[test] diff --git a/atomic-agent/src/record/mod.rs b/atomic-agent/src/record/mod.rs index 45f01373..df46caaa 100644 --- a/atomic-agent/src/record/mod.rs +++ b/atomic-agent/src/record/mod.rs @@ -90,7 +90,8 @@ use provenance::{ /// - Good prompt: `"Fix the authentication bug in login.rs"` /// - Slash command or no prompt: `"Add src/main.rs, Cargo.toml"` /// -/// The author is the agent identity. +/// The author is the agent identity — the delegated identity selected for +/// this hook invocation when one is in force, else the plus-tag fallback. fn build_turn_header( options: &TurnRecordOptions<'_>, status: &RepositoryStatus, @@ -102,6 +103,7 @@ fn build_turn_header( &options.session.agent_name, &options.session.agent_display_name, &options.session.session_id, + options.agent_identity.as_deref(), ); ChangeHeader::builder() diff --git a/atomic-agent/src/record/options.rs b/atomic-agent/src/record/options.rs index aa61685a..f4a6b11d 100644 --- a/atomic-agent/src/record/options.rs +++ b/atomic-agent/src/record/options.rs @@ -37,6 +37,17 @@ pub struct TurnRecordOptions<'a> { /// /// Used for the change message and the SessionEnvelope's prompt_summary. pub prompt: Option, + + /// Name of the delegated agent identity to sign this turn as, if one + /// is in force. + /// + /// Resolved once by the CLI hook handler (`ATOMIC_AGENT_IDENTITY` env var, + /// else the global `agent_identity` setting, else the active server + /// profile's binding) and passed as plain data — this crate never reads + /// config. `None` means no delegated identity was selected, and the + /// author falls back to the plus-tag of the default identity, exactly + /// as before agent identities existed. + pub agent_identity: Option, } /// The result of recording a turn as an Atomic change. diff --git a/atomic-agent/src/record/provenance.rs b/atomic-agent/src/record/provenance.rs index 59590a09..c1107e2c 100644 --- a/atomic-agent/src/record/provenance.rs +++ b/atomic-agent/src/record/provenance.rs @@ -244,7 +244,9 @@ pub(crate) fn build_turn_envelope( // has no delegated identity — the plus-tag path signs with the human's key // and there is no certificate to point at, which is precisely the // difference the field is there to record. - if let Some(urn) = crate::identity::active_delegation_urn(None, None) { + if let Some(urn) = + crate::identity::active_delegation_urn(options.agent_identity.as_deref(), None) + { builder = builder.delegation_id(urn); } diff --git a/atomic-agent/src/record/tests.rs b/atomic-agent/src/record/tests.rs index fe58cca1..3ed699bd 100644 --- a/atomic-agent/src/record/tests.rs +++ b/atomic-agent/src/record/tests.rs @@ -35,6 +35,7 @@ fn make_options<'a>(session: &'a AgentSession, event: &'a TurnEvent) -> TurnReco turn_number: 3, turn_duration_ms: 12400, prompt: Some("Fix the authentication bug in login.rs".to_string()), + agent_identity: None, } } @@ -869,6 +870,7 @@ fn test_record_turn_nonexistent_repo_fails() { turn_number: 3, turn_duration_ms: 5000, prompt: Some("Fix the bug".to_string()), + agent_identity: None, }; let result = record_turn(Path::new("/nonexistent/repo/path"), &options); @@ -1109,6 +1111,7 @@ fn test_orphaned_session_view_duplicates_content_on_merge() { turn_number: 1, turn_duration_ms: 1000, prompt: Some("Bump step10".to_string()), + agent_identity: None, }; record_turn(repo_root, &options_a).unwrap(); @@ -1134,6 +1137,7 @@ fn test_orphaned_session_view_duplicates_content_on_merge() { turn_number: 1, turn_duration_ms: 1000, prompt: Some("Bump step70".to_string()), + agent_identity: None, }; record_turn(repo_root, &options_b) .expect("record_turn should self-heal an orphaned session view rather than fail"); diff --git a/atomic-agent/src/turn/orchestrator/mod.rs b/atomic-agent/src/turn/orchestrator/mod.rs index 90bf19e6..1cdf88c5 100644 --- a/atomic-agent/src/turn/orchestrator/mod.rs +++ b/atomic-agent/src/turn/orchestrator/mod.rs @@ -394,6 +394,15 @@ pub struct TurnOrchestrator { /// Human-readable agent display name (e.g., "Claude Code"). pub(crate) agent_display_name: String, + /// Delegated agent identity to sign recorded turns as, if one is in + /// force for this hook invocation. + /// + /// Resolved by the CLI hook handler from the env var / global setting / + /// active server profile chain, and passed as plain data — the + /// orchestrator never reads config itself. `None` keeps the plus-tag + /// fallback, which is the behavior from before agent identities existed. + pub(crate) agent_identity: Option, + /// Managed-run context when a governing lifecycle covers this hook; /// `None` for direct agent usage (behavior unchanged). pub(crate) managed_run: Option, @@ -437,6 +446,7 @@ impl TurnOrchestrator { watcher, agent_name: "unknown".to_string(), agent_display_name: "Unknown Agent".to_string(), + agent_identity: None, managed_run: None, journal_sink: None, }) @@ -457,6 +467,7 @@ impl TurnOrchestrator { watcher, agent_name: "unknown".to_string(), agent_display_name: "Unknown Agent".to_string(), + agent_identity: None, managed_run: None, journal_sink: None, } @@ -472,6 +483,18 @@ impl TurnOrchestrator { self.agent_display_name = display_name.into(); } + /// Set the delegated agent identity that recorded turns sign as. + /// + /// Called by the CLI hook handler after construction with the name + /// resolved from the selection chain (`ATOMIC_AGENT_IDENTITY` env var, + /// else the global `agent_identity` setting, else the active server + /// profile's binding). `None` — the default — keeps the plus-tag + /// fallback. The name crosses as plain data; resolution errors were + /// already handled upstream and degrade to `None`. + pub fn set_agent_identity(&mut self, identity: Option) { + self.agent_identity = identity; + } + /// Attach the managed-run context for this hook invocation /// (see [`ManagedRunContext`]). pub fn set_managed_run(&mut self, context: ManagedRunContext) { diff --git a/atomic-agent/src/turn/orchestrator/session_end.rs b/atomic-agent/src/turn/orchestrator/session_end.rs index 9ae32b89..6b2adb34 100644 --- a/atomic-agent/src/turn/orchestrator/session_end.rs +++ b/atomic-agent/src/turn/orchestrator/session_end.rs @@ -117,6 +117,7 @@ impl TurnOrchestrator { turn_number, turn_duration_ms, prompt, + agent_identity: self.agent_identity.clone(), }; record_turn(&self.repo_root, &record_options) }; diff --git a/atomic-agent/src/turn/orchestrator/turn.rs b/atomic-agent/src/turn/orchestrator/turn.rs index 32045281..d05b5156 100644 --- a/atomic-agent/src/turn/orchestrator/turn.rs +++ b/atomic-agent/src/turn/orchestrator/turn.rs @@ -204,6 +204,7 @@ impl TurnOrchestrator { turn_number: session.turn_count.saturating_add(1), turn_duration_ms: 0, prompt: None, + agent_identity: self.agent_identity.clone(), }; if crate::record::scope::has_pending_changes(&self.repo_root, &options)? { return Err(AgentError::RecordFailed { @@ -328,6 +329,7 @@ impl TurnOrchestrator { turn_number, turn_duration_ms, prompt, + agent_identity: self.agent_identity.clone(), }; match record_turn(&self.repo_root, &record_options) { diff --git a/atomic-agent/tests/record_view_mismatch_regression.rs b/atomic-agent/tests/record_view_mismatch_regression.rs index f83a9129..6eaca5d1 100644 --- a/atomic-agent/tests/record_view_mismatch_regression.rs +++ b/atomic-agent/tests/record_view_mismatch_regression.rs @@ -30,6 +30,7 @@ fn options<'a>( turn_number, turn_duration_ms: 1000, prompt: Some(prompt.to_string()), + agent_identity: None, } } diff --git a/atomic-cli/src/commands/agent/hooks.rs b/atomic-cli/src/commands/agent/hooks.rs index 3a115235..9ebf32a6 100644 --- a/atomic-cli/src/commands/agent/hooks.rs +++ b/atomic-cli/src/commands/agent/hooks.rs @@ -264,6 +264,24 @@ impl Command for Hooks { orchestrator .set_journal_sink(Arc::new(super::owner::OwnerJournalSink::new(&repo_root))); + // The delegated identity to sign recorded turns as, resolved + // through the selection chain (env var > global setting > + // active server profile). `None` keeps the plus-tag fallback, + // so an environment with nothing configured behaves exactly as + // before agent identities existed. + let agent_identity = super::identity::resolve_effective_agent_identity(); + match &agent_identity { + Some((name, source)) => { + log::debug!("hooks: recording as agent identity '{name}' ({source})"); + } + None => { + log::debug!( + "hooks: no agent identity selected; recording as plus-tag of default" + ); + } + } + orchestrator.set_agent_identity(agent_identity.map(|(name, _)| name)); + // Under a managed lifecycle, sessions adopt the declared view // and carry the run stamp (see lifecycle module docs). if let Some(lc) = &managed { diff --git a/atomic-cli/src/commands/agent/identity.rs b/atomic-cli/src/commands/agent/identity.rs new file mode 100644 index 00000000..488f648c --- /dev/null +++ b/atomic-cli/src/commands/agent/identity.rs @@ -0,0 +1,454 @@ +//! `atomic agent identity` — select the delegated identity hooks sign as. +//! +//! Agent hooks record turns as the plus-tag of the default identity by +//! default — legible, but keyed to the human. This command tree selects a +//! delegated agent identity so every hooked agent on this machine records +//! under the agent's own key instead. +//! +//! The selection is deliberately **global** (one identity per machine, not +//! per repository): `set` writes `agent_identity` to `~/.atomic/config.toml`, +//! and every hook invocation — in any repo, for any agent — resolves it +//! fresh. `ATOMIC_AGENT_IDENTITY` remains the per-process escape hatch and +//! takes precedence, and the active server profile's `agent_identity` +//! binding (written by `atomic identity agent create`) is the implicit +//! fallback when nothing explicit is set. +//! +//! Resolution order (shared with the hook handler, see +//! [`resolve_effective_agent_identity`]): +//! +//! ```text +//! ATOMIC_AGENT_IDENTITY env var +//! > global agent_identity setting +//! > active server profile's agent_identity +//! > plus-tag fallback (behavior from before agent identities existed) +//! ``` +//! +//! # Examples +//! +//! ```text +//! # Select an agent identity (machine-wide) +//! atomic agent identity set fred+opencode +//! +//! # Stop recording as the agent +//! atomic agent identity unset +//! +//! # See what hooks will sign as, and why +//! atomic agent identity show +//! ``` + +use clap::{Args, Subcommand}; + +use atomic_canonical::delegation as cert; +use atomic_identity::IdentityStore; + +use atomic_config::GlobalConfig; + +use crate::commands::Command; +use crate::error::{CliError, CliResult}; +use crate::output::{print_hint, print_success, print_warning}; + +// Identity Command + +/// Select or inspect the agent identity that recording hooks sign as. +#[derive(Debug, Args)] +#[command(arg_required_else_help = true)] +pub struct Identity { + #[command(subcommand)] + command: IdentityCommands, +} + +/// Available `atomic agent identity` subcommands. +#[derive(Debug, Subcommand)] +pub enum IdentityCommands { + /// Select the agent identity for hook recording (machine-wide). + /// + /// Validates the identity eagerly — it must exist in the identity store + /// and be an agent/delegated identity, since a human identity here + /// would silently sign agent work with the human's key — then writes it + /// to `~/.atomic/config.toml` as the global `agent_identity`. Every + /// hooked agent on this machine records under it from the next hook + /// invocation on. + /// + /// # Examples + /// + /// ```text + /// atomic agent identity set fred+opencode + /// ``` + Set(Set), + + /// Clear the agent identity selection. + /// + /// Hooks go back to recording as the plus-tag of the default identity — + /// the behavior from before agent identities existed. + /// + /// # Examples + /// + /// ```text + /// atomic agent identity unset + /// ``` + Unset(Unset), + + /// Show the effective agent identity and where it comes from. + /// + /// Runs the same resolution the hooks use (env var, global setting, + /// active server profile, or none), so what you see here is exactly + /// what the next recorded turn will sign as. + /// + /// # Examples + /// + /// ```text + /// atomic agent identity show + /// ``` + Show(Show), +} + +impl Command for Identity { + fn run(&self) -> CliResult<()> { + match &self.command { + IdentityCommands::Set(cmd) => cmd.run(), + IdentityCommands::Unset(cmd) => cmd.run(), + IdentityCommands::Show(cmd) => cmd.run(), + } + } +} + +// Effective Identity Resolution + +/// Where a resolved agent identity came from. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub enum AgentIdentitySource { + /// The `ATOMIC_AGENT_IDENTITY` environment variable. + EnvVar, + /// The global `agent_identity` setting in `~/.atomic/config.toml`. + GlobalSetting, + /// The active server profile's `agent_identity` binding. + ServerProfile, +} + +impl std::fmt::Display for AgentIdentitySource { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + f.write_str(match self { + AgentIdentitySource::EnvVar => "ATOMIC_AGENT_IDENTITY env var", + AgentIdentitySource::GlobalSetting => "global setting (~/.atomic/config.toml)", + AgentIdentitySource::ServerProfile => "active server profile", + }) + } +} + +impl AgentIdentitySource { + /// Stable machine key for JSON output (`env`, `global`, `server-profile`). + /// The human labels live in the `Display` impl. + pub fn key(&self) -> &'static str { + match self { + AgentIdentitySource::EnvVar => "env", + AgentIdentitySource::GlobalSetting => "global", + AgentIdentitySource::ServerProfile => "server-profile", + } + } +} + +/// Resolve the agent identity that recording hooks should sign as. +/// +/// Order: `ATOMIC_AGENT_IDENTITY` env var, else the global `agent_identity` +/// setting, else the active server profile's `agent_identity` binding. +/// Returns `None` when nothing is configured — hooks then record as the +/// plus-tag of the default identity, exactly as before agent identities +/// existed. Config read failures degrade to `None`: a hook must never fail +/// over identity selection. +pub fn resolve_effective_agent_identity() -> Option<(String, AgentIdentitySource)> { + // 1. Env var — the per-process escape hatch, and the only handle a + // remote runner has. + let env = std::env::var(atomic_agent::identity::AGENT_IDENTITY_ENV) + .ok() + .map(|v| v.trim().to_string()) + .filter(|v| !v.is_empty()); + + // An unreadable config degrades to no selection rather than failing + // the hook. + let config = GlobalConfig::load().ok()?; + resolve_from_parts(env.as_deref(), &config) +} + +/// Pure resolution over the chain, split from the I/O so the order is +/// unit-testable. +/// +/// Env var beats the global setting beats the active server profile; empty +/// or whitespace-only values are skipped as if unset at every level. +fn resolve_from_parts( + env: Option<&str>, + config: &GlobalConfig, +) -> Option<(String, AgentIdentitySource)> { + if let Some(name) = env.map(str::trim).filter(|n| !n.is_empty()) { + return Some((name.to_string(), AgentIdentitySource::EnvVar)); + } + if let Some(name) = config.agent_identity.as_deref().map(str::trim) { + if !name.is_empty() { + return Some((name.to_string(), AgentIdentitySource::GlobalSetting)); + } + } + if let Some(name) = config.active_server_agent_identity().map(str::trim) { + if !name.is_empty() { + return Some((name.to_string(), AgentIdentitySource::ServerProfile)); + } + } + None +} + +// Set Command + +/// Select the agent identity for hook recording. +#[derive(Debug, Args)] +pub struct Set { + /// Name of the agent identity to sign as (e.g., "fred+opencode"). + name: String, +} + +impl Command for Set { + fn run(&self) -> CliResult<()> { + let name = self.name.trim(); + if name.is_empty() { + return Err(CliError::InvalidArgument { + message: "Identity name cannot be empty".to_string(), + }); + } + + // Eager validation: the record path falls back softly on a bad + // name, so this is the one place with room to say what's wrong. + let store = IdentityStore::open_default().map_err(|e| { + CliError::Internal(anyhow::anyhow!("Failed to open identity store: {e}")) + })?; + let identity = store + .load_by_name(name) + .map_err(|_| CliError::IdentityNotFound(name.to_string()))?; + + // A human identity here would silently sign agent work as the + // human with no delegation behind it — refuse it outright. + if !identity.identity_type.is_delegated() && !identity.identity_type.is_agent() { + return Err(CliError::InvalidArgument { + message: format!( + "'{name}' is a human identity. Agent hooks can only sign as an agent \ + identity — create one with `atomic identity agent create`." + ), + }); + } + + // Keyed attribution without a certificate still records (the key is + // the agent's), but the envelope will carry no delegation URN and a + // server cannot confirm the delegation. That is worth saying now, + // not discovering from a change header later. + if cert::active_for_delegate(&store, &identity).is_none() { + print_warning(&format!( + "No active delegation certificate for '{name}'. Turns will record keyed to \ + the agent, but the envelope will carry no delegation URN until a \ + certificate is granted (`atomic identity grant new {name}`)." + )); + } + + let mut config = GlobalConfig::load().map_err(|e| { + CliError::Internal(anyhow::anyhow!("Failed to load global config: {e}")) + })?; + config.agent_identity = Some(name.to_string()); + config.save().map_err(|e| { + CliError::Internal(anyhow::anyhow!("Failed to save global config: {e}")) + })?; + + print_success(&format!( + "Agent identity set to '{name}' — recording hooks now sign as it" + )); + print_hint("Effective immediately for every hooked agent on this machine"); + Ok(()) + } +} + +// Unset Command + +/// Clear the agent identity selection. +#[derive(Debug, Args)] +pub struct Unset {} + +impl Command for Unset { + fn run(&self) -> CliResult<()> { + let mut config = GlobalConfig::load().map_err(|e| { + CliError::Internal(anyhow::anyhow!("Failed to load global config: {e}")) + })?; + let had = config.agent_identity.is_some(); + config.agent_identity = None; + config.save().map_err(|e| { + CliError::Internal(anyhow::anyhow!("Failed to save global config: {e}")) + })?; + + if had { + print_success( + "Agent identity unset — hooks record as the plus-tag of the default identity", + ); + } else { + print_hint("No agent identity was set"); + } + Ok(()) + } +} + +// Show Command + +/// Show the effective agent identity and where it comes from. +#[derive(Debug, Args)] +pub struct Show {} + +impl Command for Show { + fn run(&self) -> CliResult<()> { + match resolve_effective_agent_identity() { + Some((name, source)) => { + print_success(&format!("Agent identity: {name}")); + println!(" Source: {source}"); + // The record path falls back softly when the name does not + // resolve — show should say so, because that is a silent + // difference between what is selected and what records. + match IdentityStore::open_default() { + Ok(store) => match store.load_by_name(&name) { + Ok(identity) => { + println!(" Email: {}", identity.email.as_deref().unwrap_or("-")); + println!(" Key: {}", identity.public_key_base32()); + match cert::active_for_delegate(&store, &identity) { + Some(d) => { + println!(" Delegation: {}", d.delegation.id.to_urn()); + } + None => { + print_warning( + "No active delegation certificate — turns record \ + keyed to the agent but carry no delegation URN", + ); + } + } + } + Err(_) => { + print_warning(&format!( + "'{name}' does not resolve in the identity store — \ + hooks will fall back to the plus-tag path" + )); + } + }, + Err(e) => { + print_warning(&format!( + "Could not open identity store ({e}) — hooks will fall \ + back to the plus-tag path if the identity stays unresolvable" + )); + } + } + } + None => { + print_hint( + "No agent identity selected — recording hooks sign as the \ + plus-tag of the default identity", + ); + print_hint("Select one with: atomic agent identity set "); + } + } + Ok(()) + } +} + +// Tests + +#[cfg(test)] +mod tests { + use super::*; + + // The env var and the real ~/.atomic/config.toml both sit outside this + // test's control, so the chain is pinned against synthetic parts via + // resolve_from_parts — the same function the hook handler reaches + // through resolve_effective_agent_identity. + #[test] + fn env_var_beats_everything() { + let config = GlobalConfig { + agent_identity: Some("global+agent".to_string()), + server: atomic_config::ServerConfig { + agent_identity: Some("server+agent".to_string()), + ..atomic_config::ServerConfig::default() + }, + ..GlobalConfig::default() + }; + let (name, source) = + resolve_from_parts(Some("env+agent"), &config).expect("env var must win"); + assert_eq!(name, "env+agent"); + assert_eq!(source, AgentIdentitySource::EnvVar); + } + + #[test] + fn global_setting_beats_server_profile() { + let config = GlobalConfig { + agent_identity: Some("global+agent".to_string()), + server: atomic_config::ServerConfig { + agent_identity: Some("server+agent".to_string()), + ..atomic_config::ServerConfig::default() + }, + ..GlobalConfig::default() + }; + let (name, source) = + resolve_from_parts(None, &config).expect("global setting must beat profile"); + assert_eq!(name, "global+agent"); + assert_eq!(source, AgentIdentitySource::GlobalSetting); + } + + #[test] + fn server_profile_is_the_implicit_fallback() { + let config = GlobalConfig { + server: atomic_config::ServerConfig { + agent_identity: Some("server+agent".to_string()), + ..atomic_config::ServerConfig::default() + }, + ..GlobalConfig::default() + }; + let (name, source) = + resolve_from_parts(None, &config).expect("profile binding must be used"); + assert_eq!(name, "server+agent"); + assert_eq!(source, AgentIdentitySource::ServerProfile); + } + + #[test] + fn nothing_set_means_no_selection() { + // The backward-compatibility guarantee: with nothing configured, + // there is no selection and hooks keep the plus-tag path. + assert!(resolve_from_parts(None, &GlobalConfig::default()).is_none()); + } + + #[test] + fn blank_values_are_skipped_at_every_level() { + let config = GlobalConfig { + agent_identity: Some(" ".to_string()), + server: atomic_config::ServerConfig { + agent_identity: Some("".to_string()), + ..atomic_config::ServerConfig::default() + }, + ..GlobalConfig::default() + }; + assert!(resolve_from_parts(Some(" "), &config).is_none()); + + // A blank global falls through to the profile binding. + let config = GlobalConfig { + server: atomic_config::ServerConfig { + agent_identity: Some("server+agent".to_string()), + ..atomic_config::ServerConfig::default() + }, + ..GlobalConfig::default() + }; + let (name, source) = resolve_from_parts(None, &config).unwrap(); + assert_eq!(name, "server+agent"); + assert_eq!(source, AgentIdentitySource::ServerProfile); + } + + #[test] + fn source_labels_are_stable() { + assert_eq!( + AgentIdentitySource::EnvVar.to_string(), + "ATOMIC_AGENT_IDENTITY env var" + ); + assert_eq!( + AgentIdentitySource::GlobalSetting.to_string(), + "global setting (~/.atomic/config.toml)" + ); + assert_eq!( + AgentIdentitySource::ServerProfile.to_string(), + "active server profile" + ); + } +} diff --git a/atomic-cli/src/commands/agent/mod.rs b/atomic-cli/src/commands/agent/mod.rs index 67fa5202..dc4df1ea 100644 --- a/atomic-cli/src/commands/agent/mod.rs +++ b/atomic-cli/src/commands/agent/mod.rs @@ -42,6 +42,7 @@ mod disable; mod enable; mod explain; mod hooks; +mod identity; mod lifecycle; mod owner; mod status; @@ -55,6 +56,7 @@ pub use attest::Attest; pub use disable::Disable; pub use enable::Enable; pub use explain::Explain; +pub use identity::Identity; pub use lifecycle::Lifecycle; pub use status::AgentStatus; @@ -206,6 +208,29 @@ pub enum AgentCommands { #[command(name = "database-owner", hide = true)] DatabaseOwner(owner::DatabaseOwner), + /// Select or inspect the agent identity that recording hooks sign as. + /// + /// `set` makes every hooked agent on this machine record under a + /// delegated agent identity's own key instead of the plus-tag of the + /// default identity. The selection is global — one identity per + /// machine — with `ATOMIC_AGENT_IDENTITY` as the per-process escape + /// hatch and the active server profile's `agent_identity` binding as + /// the implicit fallback. + /// + /// # Examples + /// + /// ```text + /// # Select an agent identity + /// atomic agent identity set fred+opencode + /// + /// # See what hooks will sign as, and why + /// atomic agent identity show + /// + /// # Back to the plus-tag fallback + /// atomic agent identity unset + /// ``` + Identity(Identity), + /// Internal hook handlers (called by agent hooks). /// /// These commands are invoked by the hooks installed in agent @@ -228,6 +253,7 @@ impl Command for Agent { AgentCommands::Attest(cmd) => cmd.run(), AgentCommands::Lifecycle(cmd) => cmd.run(), AgentCommands::DatabaseOwner(cmd) => cmd.run(), + AgentCommands::Identity(cmd) => cmd.run(), AgentCommands::Hooks(cmd) => cmd.run(), } } diff --git a/atomic-cli/src/commands/agent/status.rs b/atomic-cli/src/commands/agent/status.rs index 4a9a2540..498582b7 100644 --- a/atomic-cli/src/commands/agent/status.rs +++ b/atomic-cli/src/commands/agent/status.rs @@ -32,7 +32,8 @@ use crate::error::CliResult; /// Show agent integration status. /// /// Displays which agents have hooks installed, any active sessions, -/// and the current state of the file watcher. +/// the current state of the file watcher, and the delegated identity +/// recording hooks sign as. #[derive(Debug, Args)] pub struct AgentStatus { /// Show detailed session information. @@ -95,6 +96,12 @@ struct StatusJson { agents: Vec, sessions: Vec, totals: Totals, + /// The delegated identity hooks sign as, when one is selected. Absent + /// when hooks record as the plus-tag of the default identity — the + /// same resolution the hooks themselves run, so what this reports is + /// what the next recorded turn will use. + #[serde(skip_serializing_if = "Option::is_none")] + agent_identity: Option, /// Why the session list is empty, when it is empty for a reason. The human /// output prints this and carries on; dropping it from JSON would turn a /// broken session store into an indistinguishable "no sessions". @@ -102,6 +109,17 @@ struct StatusJson { sessions_error: Option, } +/// The effective agent identity, as data. +#[derive(Debug, Serialize)] +struct AgentIdentityJson { + /// The identity's name in the store, e.g. `fred+opencode`. + name: String, + /// Where the selection came from — one of `env`, `global`, + /// `server-profile`. Stable keys; the human labels live in the + /// `agent identity` command. + source: &'static str, +} + impl AgentStatus { /// Create a default instance for testing. #[cfg(test)] @@ -167,6 +185,12 @@ impl AgentStatus { }) .collect(), totals, + agent_identity: super::identity::resolve_effective_agent_identity().map( + |(name, source)| AgentIdentityJson { + name, + source: source.key(), + }, + ), sessions_error, } } @@ -193,6 +217,19 @@ impl Command for AgentStatus { println!("======================="); println!(); + // Identity — the delegated identity hooks sign as, machine-global, + // shown even when no hooks are installed yet because the selection + // is exactly what a future `atomic agent enable` will record as. + match super::identity::resolve_effective_agent_identity() { + Some((name, source)) => { + println!(" Identity: {name} ({source})"); + } + None => { + println!(" Identity: none — hooks record as the plus-tag of the default identity"); + } + } + println!(); + let installed = registry.installed(&repo_root); let detected = registry.detect(&repo_root); diff --git a/atomic-cli/tests/database_owner_integration_test.rs b/atomic-cli/tests/database_owner_integration_test.rs index 28d61d81..6a4b15c8 100644 --- a/atomic-cli/tests/database_owner_integration_test.rs +++ b/atomic-cli/tests/database_owner_integration_test.rs @@ -12,6 +12,18 @@ use fs2::FileExt; use serde_json::Value; use tempfile::TempDir; +// Every test in this file spawns real `atomic` subprocesses that hold the +// redb lock and burn CPU while they run. On 2-core Windows CI runners the +// default test-threads parallelism lets the process-heavy tests (the +// eight-process session-start hammer, the failpoint owners) starve whichever +// sibling tests happen to overlap with them past their database-wait budgets +// — the failures move between runs (concurrent_session_starts…, +// crashing_second_checkpoint…, owner_death_after_checkpoint…, all in the +// same file), which is the signature of scheduler starvation, not a bug in +// any one test. Serializing the file trades a few minutes of wall time for +// runs that fail only when something is actually broken. +use serial_test::serial; + // Bound child processes so a platform-specific IPC regression produces a // useful failure instead of occupying a CI runner indefinitely. Drain output // concurrently so a full pipe cannot prevent the child from exiting. @@ -198,6 +210,7 @@ fn run_agent_hook(repository: &std::path::Path, agent: &str, verb: &str, payload } #[test] +#[serial] fn concurrent_session_starts_wait_for_writer_and_persist_views_and_lifecycle() { let temp = TempDir::new().unwrap(); let repository = temp.path().join("repo"); @@ -250,6 +263,7 @@ fn concurrent_session_starts_wait_for_writer_and_persist_views_and_lifecycle() { } #[test] +#[serial] fn read_only_turn_does_not_reuse_its_goal_for_the_next_turn_across_agents() { for (agent, prompt_verb) in [ ("codex", "user-prompt-submit"), @@ -355,6 +369,7 @@ fn wait_for_shutdown(repository: &std::path::Path) { } #[test] +#[serial] fn concurrent_hooks_commit_lossless_envelopes_without_output_or_drops() { const TOOL_EVENTS: usize = 16; @@ -498,6 +513,7 @@ fn concurrent_hooks_commit_lossless_envelopes_without_output_or_drops() { } #[test] +#[serial] fn lifecycle_stop_resume_zombie_abandon_and_lease_expiry() { let temp = TempDir::new().unwrap(); let repository = temp.path().join("repo"); @@ -726,6 +742,7 @@ fn lifecycle_stop_resume_zombie_abandon_and_lease_expiry() { } #[test] +#[serial] fn owner_death_before_and_after_event_commit_retries_exactly_once() { for failpoint in ["before-envelope-commit", "after-envelope-commit"] { let temp = TempDir::new().unwrap(); @@ -803,6 +820,7 @@ fn batch_owner_rpc(repository: &std::path::Path, request: &Value) -> std::io::Re #[cfg(unix)] #[test] +#[serial] fn owner_batch_crash_retries_the_whole_durable_batch_exactly_once() { for failpoint in ["before-envelope-commit", "after-envelope-commit"] { let temp = TempDir::new().unwrap(); @@ -865,6 +883,7 @@ fn owner_batch_crash_retries_the_whole_durable_batch_exactly_once() { } #[test] +#[serial] fn owner_death_after_checkpoint_prepare_and_bind_recovers_in_hook_process() { for failpoint in ["after-checkpoint-prepare", "after-checkpoint-bind"] { let temp = TempDir::new().unwrap(); @@ -929,6 +948,7 @@ fn owner_death_after_checkpoint_prepare_and_bind_recovers_in_hook_process() { } #[test] +#[serial] fn corrupt_legacy_graph_is_retained_when_migration_cannot_verify_it() { let temp = TempDir::new().unwrap(); let repository = temp.path().join("repo"); @@ -962,6 +982,7 @@ fn corrupt_legacy_graph_is_retained_when_migration_cannot_verify_it() { } #[test] +#[serial] fn legacy_graph_pending_delta_imports_once_then_json_authority_is_removed() { let temp = TempDir::new().unwrap(); let repository = temp.path().join("repo"); @@ -1083,6 +1104,7 @@ fn legacy_graph_pending_delta_imports_once_then_json_authority_is_removed() { } #[test] +#[serial] fn turn_end_publishes_one_checkpoint_turn_and_advances_head() { let temp = TempDir::new().unwrap(); let repository = temp.path().join("repo"); @@ -1164,6 +1186,7 @@ fn turn_end_publishes_one_checkpoint_turn_and_advances_head() { } #[test] +#[serial] fn concurrent_stops_publish_ordered_ledgers_without_external_serialization() { let temp = TempDir::new().unwrap(); let repository = temp.path().join("repo"); @@ -1295,6 +1318,7 @@ fn concurrent_stops_publish_ordered_ledgers_without_external_serialization() { } #[test] +#[serial] fn stop_publication_timeout_preserves_the_active_turn_for_retry() { let temp = TempDir::new().unwrap(); let repository = temp.path().join("repo"); @@ -1337,6 +1361,7 @@ fn stop_publication_timeout_preserves_the_active_turn_for_retry() { } #[test] +#[serial] fn killed_stop_releases_publication_lock_and_can_be_retried() { let temp = TempDir::new().unwrap(); let repository = temp.path().join("repo"); @@ -1568,6 +1593,7 @@ fn large_checkpoint_fixture( } #[test] +#[serial] fn large_frozen_checkpoint_completes_across_agents() { for agent in ["codex", "claude-code", "opencode"] { large_checkpoint_fixture(agent, 1024, 1024, false); @@ -1575,11 +1601,13 @@ fn large_frozen_checkpoint_completes_across_agents() { } #[test] +#[serial] fn large_envelope_checkpoint_recovers_between_pages() { large_checkpoint_fixture("claude-code", 1, 600 * 1024, true); } #[test] +#[serial] fn failed_frozen_read_resumes_recorded_changes_on_next_stop() { let temp = TempDir::new().unwrap(); let repository = temp.path().join("repo"); @@ -1658,6 +1686,7 @@ fn failed_frozen_read_resumes_recorded_changes_on_next_stop() { } #[test] +#[serial] fn pre_cutover_session_count_gap_publishes_at_next_ledger_ordinal() { let temp = TempDir::new().unwrap(); let repository = temp.path().join("repo"); @@ -1704,6 +1733,7 @@ fn pre_cutover_session_count_gap_publishes_at_next_ledger_ordinal() { } #[test] +#[serial] fn crashing_second_checkpoint_keeps_first_turn_immutable() { let temp = TempDir::new().unwrap(); let repository = temp.path().join("repo"); @@ -1773,6 +1803,7 @@ fn crashing_second_checkpoint_keeps_first_turn_immutable() { } #[test] +#[serial] fn owner_election_commit_reconnect_and_crash_restart() { let temp = TempDir::new().unwrap(); let repository = temp.path().join("repo"); @@ -1834,6 +1865,7 @@ fn owner_election_commit_reconnect_and_crash_restart() { } #[test] +#[serial] fn scoped_concurrent_stops_and_session_end_do_not_record_sibling_files() { let temp = TempDir::new().unwrap(); let root = temp.path().join("repo"); @@ -2025,16 +2057,19 @@ fn check_missing_turn_start(write_files: bool) { } #[test] +#[serial] fn missing_turn_start_recovers_changes_and_provenance_without_duplicate_stops() { check_missing_turn_start(true); } #[test] +#[serial] fn missing_turn_start_recovers_read_only_provenance() { check_missing_turn_start(false); } #[test] +#[serial] fn scoped_stop_without_start_or_journal_fails_visibly_and_can_retry() { let temp = TempDir::new().unwrap(); let root = temp.path().join("repo"); diff --git a/atomic-config/src/lib.rs b/atomic-config/src/lib.rs index b39a3cb2..0b463aa0 100644 --- a/atomic-config/src/lib.rs +++ b/atomic-config/src/lib.rs @@ -240,6 +240,19 @@ pub struct GlobalConfig { /// When `None`, the legacy `[server]` block is used. #[serde(default, skip_serializing_if = "Option::is_none")] pub default_server: Option, + + /// Machine-wide agent identity that recording hooks sign as. + /// + /// Set by `atomic agent identity set `. When hooks record a + /// turn, the change is attributed to this identity's own key instead + /// of falling back to the plus-tag of the default identity. Deliberately + /// global — one agent identity per machine, not per repository — with + /// `ATOMIC_AGENT_IDENTITY` as the per-process escape hatch and the + /// active server profile's `agent_identity` as the implicit fallback. + /// + /// Example: `agent_identity = "fred+opencode"` + #[serde(default, skip_serializing_if = "Option::is_none")] + pub agent_identity: Option, } fn default_channel_name() -> String { @@ -257,6 +270,7 @@ impl Default for GlobalConfig { server: ServerConfig::default(), servers: BTreeMap::new(), default_server: None, + agent_identity: None, } } } @@ -350,6 +364,22 @@ impl GlobalConfig { None => Ok((&mut self.server, None)), } } + + /// The active server profile's `agent_identity` binding, if any. + /// + /// Active-profile resolution mirrors [`resolve_server`](Self::resolve_server) + /// with no override: `default_server` → named profile, else the legacy + /// `[server]` block. Unlike `resolve_server`, a dangling `default_server` + /// degrades to the legacy block rather than erroring — this is a fallback + /// link in the recording hook's identity chain, and recording must never + /// fail over a misconfigured profile name. + pub fn active_server_agent_identity(&self) -> Option<&str> { + let profile = match self.default_server.as_deref() { + Some(name) => self.servers.get(name).unwrap_or(&self.server), + None => &self.server, + }; + profile.agent_identity.as_deref() + } } /// Color output preference @@ -914,6 +944,69 @@ default_org = "alice" ); } + #[test] + fn test_agent_identity_roundtrip() { + let config = GlobalConfig { + agent_identity: Some("fred+opencode".to_string()), + ..GlobalConfig::default() + }; + + let toml_str = toml::to_string_pretty(&config).unwrap(); + assert!(toml_str.contains("agent_identity = \"fred+opencode\"")); + + let parsed: GlobalConfig = toml::from_str(&toml_str).unwrap(); + assert_eq!(parsed.agent_identity.as_deref(), Some("fred+opencode")); + } + + #[test] + fn test_agent_identity_absent_in_legacy_configs() { + // Configs written before the field existed must deserialize with + // no selection — hooks fall back to the plus-tag path unchanged. + let legacy = "default_channel = \"main\"\n"; + let parsed: GlobalConfig = toml::from_str(legacy).unwrap(); + assert_eq!(parsed.agent_identity, None); + } + + #[test] + fn test_agent_identity_omitted_when_unset() { + // skip_serializing_if: an unset selection leaves no key behind, + // so hand-edited configs stay clean. + let toml_str = toml::to_string_pretty(&GlobalConfig::default()).unwrap(); + assert!(!toml_str.contains("agent_identity")); + } + + #[test] + fn test_active_server_agent_identity_prefers_default_profile() { + let mut config = GlobalConfig::default(); + config.server.agent_identity = Some("legacy+agent".to_string()); + assert_eq!(config.active_server_agent_identity(), Some("legacy+agent")); + + let named = ServerConfig { + agent_identity: Some("named+agent".to_string()), + ..ServerConfig::default() + }; + config.servers.insert("prod".to_string(), named); + config.default_server = Some("prod".to_string()); + assert_eq!(config.active_server_agent_identity(), Some("named+agent")); + } + + #[test] + fn test_active_server_agent_identity_degrades_on_dangling_default() { + // A default_server pointing at a removed profile must not break + // resolution — the legacy block is the honest active profile then. + let mut config = GlobalConfig::default(); + config.server.agent_identity = Some("legacy+agent".to_string()); + config.default_server = Some("gone".to_string()); + assert_eq!(config.active_server_agent_identity(), Some("legacy+agent")); + } + + #[test] + fn test_active_server_agent_identity_none_when_unbound() { + let mut config = GlobalConfig::default(); + config.server.agent_identity = None; + assert_eq!(config.active_server_agent_identity(), None); + } + #[test] fn global_config_dir_uses_env_override_verbatim() { // With the override set, the directory is used exactly as given. From 511ab8fb198fd835e00f19062598c2f5b7eb03c1 Mon Sep 17 00:00:00 2001 From: Bradley Hilton Date: Fri, 25 Sep 2026 14:10:45 -0700 Subject: [PATCH 16/22] fix(triage): default to the current view and --into to its parent (#217) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit `triage review` and `triage candidates` required both a source view and `--into`, so the whole command had to be retyped from memory on every invocation. The common gesture — "is the view I'm working on ready to promote?" — is now a bare `atomic triage review`. - `` is optional and defaults to the current view; the metavar drops `FEATURE` for `VIEW`, since `feature` read as a literal view name. - `--into` is optional and defaults to the view's *direct* parent, matching bare `atomic insert`. Deliberately not `nearest_shared_ancestor`, so feature-login -> service-auth -> dev targets service-auth. - Both args get view-name shell completion, which neither had. `--help` also showed zero examples: the `# Examples` doc blocks became `long_about`, which `apply_agent_help` strips tree-wide. Moved to `after_help` — the one slot the agent template preserves — with `--walkthrough` led on, as it was the undiscoverable flag. An unknown view name now explains itself instead of dead-ending on "not found", naming the current view as the value to drop in or omit: No view named 'feature' — takes a real view name, not a placeholder. Current view is 'feature-x'; omit to triage that instead, or run 'atomic view list' to see all views. Resolution goes through the cheap parent_change_count() rather than get_view_info(), which materialises the parent's whole visible change set. --- atomic-cli/src/commands/triage/mod.rs | 351 +++++++++++++++++++++++--- atomic-cli/src/main.rs | 18 +- 2 files changed, 318 insertions(+), 51 deletions(-) diff --git a/atomic-cli/src/commands/triage/mod.rs b/atomic-cli/src/commands/triage/mod.rs index e82490ed..2a9aa217 100644 --- a/atomic-cli/src/commands/triage/mod.rs +++ b/atomic-cli/src/commands/triage/mod.rs @@ -9,9 +9,11 @@ use std::path::{Path, PathBuf}; use std::process::{Command as ProcessCommand, Stdio}; use clap::{Parser, Subcommand}; +use clap_complete::engine::ArgValueCompleter; -use atomic_repository::Repository; +use atomic_repository::{Repository, RepositoryError}; +use crate::commands::complete::complete_view_names; use crate::commands::{find_repository_root, Command}; use crate::error::{CliError, CliResult}; @@ -20,7 +22,15 @@ pub mod output; pub mod project; /// Triage a feature view against a target before insert. +/// +/// Both subcommands take an optional source view and an optional `--into` +/// target: the source defaults to the current view and the target to that +/// view's parent, so a bare `atomic triage review` answers "is what I am +/// working on ready to promote?". #[derive(Debug, Parser)] +#[command(after_help = "\ +The argument and --into are both optional. Run 'atomic triage review +--help' for the full example set.")] pub struct Triage { #[command(subcommand)] pub command: TriageCommands, @@ -38,45 +48,50 @@ impl Command for Triage { /// Subcommands for `atomic triage`. #[derive(Subcommand, Debug)] pub enum TriageCommands { - /// Report the candidate change set of a feature view relative to a target. + /// Report the candidate change set of a view relative to a target. /// - /// # Examples - /// - /// ```text - /// atomic triage candidates feature --into dev - /// atomic triage candidates feature --into dev --json - /// ``` + /// Which changes would land if this view were promoted: only-in-source, + /// the transitive dependency-closure additions, and which additions are + /// "baggage" (covered by no intent). See this subcommand's examples. Candidates(TriageCandidates), /// Build the canonical triage report and render it (verdict + findings). /// /// Walks the change → file → task → intent → acceptance-criterion join, /// gates each reached intent, and emits a bounded CLI dashboard (default) - /// or the full JSON worklist (`--json`). - /// - /// # Examples - /// - /// ```text - /// atomic triage review feature --into dev - /// atomic triage review feature --into dev --json - /// atomic triage review feature --into dev --walkthrough # guided reading order - /// atomic triage review feature --into dev --html # write + open in browser - /// atomic triage review feature --into dev --html --output review.html - /// atomic triage review feature --into dev --html --no-open - /// atomic triage review feature --into dev --attest > review.signed.json - /// ``` + /// or the full JSON worklist (`--json`). See this subcommand's examples. Review(TriageReview), } -/// Compute the triage candidate set for `` relative to `--into`. +/// Compute the triage candidate set for a view relative to a target. #[derive(Debug, Parser)] -pub struct TriageCandidates { - /// The feature (source) view to review. - pub feature: String, +#[command(after_help = "\ + defaults to the current view; --into defaults to that view's parent. - /// The target view the feature would be inserted into. - #[arg(long)] - pub into: String, +Examples: + # What would land if the current view were promoted? (no arguments needed) + atomic triage candidates + atomic triage candidates --json + + # A specific source and target + atomic triage candidates my-feature --into dev + atomic triage candidates my-feature --into dev --json")] +pub struct TriageCandidates { + /// The feature (source) view to review. Defaults to the current view. + #[arg( + value_name = "VIEW", + add = ArgValueCompleter::new(complete_view_names) + )] + pub feature: Option, + + /// The target view the feature would be inserted into. Defaults to the + /// feature view's parent. + #[arg( + long, + value_name = "VIEW", + add = ArgValueCompleter::new(complete_view_names) + )] + pub into: Option, /// Emit the candidate set as JSON. #[arg(long)] @@ -87,9 +102,10 @@ impl Command for TriageCandidates { fn run(&self) -> CliResult<()> { let root = find_repository_root()?; let repo = Repository::open(&root).map_err(CliError::Repository)?; + let (feature, into) = resolve_views(&repo, self.feature.as_deref(), self.into.as_deref())?; let set = repo - .triage_candidate_set(&self.feature, &self.into) + .triage_candidate_set(&feature, &into) .map_err(CliError::Repository)?; if self.json { @@ -203,15 +219,50 @@ fn open_in_browser(target: &str) -> std::io::Result<()> { Ok(()) } -/// Build the canonical triage report for `` relative to `--into`. +/// Build the canonical triage report for a view relative to a target. #[derive(Debug, Parser)] -pub struct TriageReview { - /// The feature (source) view to review. - pub feature: String, +#[command(after_help = "\ + defaults to the current view; --into defaults to that view's parent. So a +bare 'atomic triage review' asks: is the view I am working on ready to promote? - /// The target view the feature would be inserted into. - #[arg(long)] - pub into: String, +Examples: + # Promote-readiness of the current view, in guided reading order + atomic triage review --walkthrough + + # Bounded dashboard: verdict + findings, for the current view + atomic triage review + + # The full JSON worklist — start here when driving this by hand + atomic triage review --json + + # A specific source and target + atomic triage review my-feature --into dev + atomic triage review my-feature --into dev --json + atomic triage review my-feature --into dev --walkthrough + + # Chapter tour in a browser + atomic triage review my-feature --into dev --html + atomic triage review my-feature --into dev --html --output review.html + atomic triage review my-feature --into dev --html --no-open + + # Signed export for portability/compliance + atomic triage review my-feature --into dev --attest > review.signed.json")] +pub struct TriageReview { + /// The feature (source) view to review. Defaults to the current view. + #[arg( + value_name = "VIEW", + add = ArgValueCompleter::new(complete_view_names) + )] + pub feature: Option, + + /// The target view the feature would be inserted into. Defaults to the + /// feature view's parent. + #[arg( + long, + value_name = "VIEW", + add = ArgValueCompleter::new(complete_view_names) + )] + pub into: Option, /// Emit the full report as JSON instead of the bounded CLI dashboard. #[arg(long)] @@ -251,8 +302,9 @@ impl Command for TriageReview { fn run(&self) -> CliResult<()> { let root = find_repository_root()?; let repo = Repository::open(&root).map_err(CliError::Repository)?; + let (feature, into) = resolve_views(&repo, self.feature.as_deref(), self.into.as_deref())?; - let report = project::build_report(&repo, &self.feature, &self.into)?; + let report = project::build_report(&repo, &feature, &into)?; // Output selection precedence: // attest > html > json > walkthrough > CLI dashboard. @@ -273,3 +325,228 @@ impl Command for TriageReview { Ok(()) } } + +/// Resolve the `(source, target)` view pair a triage run operates on. +/// +/// Both arguments are optional so the common gesture needs no arguments at all: +/// the source is the current view and the target is that view's direct parent +/// — the same default `atomic insert` uses for a bare promote. `arg` names the +/// argument a name came from so an unknown view can be reported against it. +/// +/// # Errors +/// +/// Returns [`CliError::InvalidArgument`] if an explicitly named view does not +/// exist, or if the target was omitted and the source is a root view (nothing +/// to promote into). +fn resolve_views( + repo: &Repository, + feature: Option<&str>, + into: Option<&str>, +) -> CliResult<(String, String)> { + let source = match feature { + Some(name) => { + require_view(repo, name, "")?; + name.to_string() + } + None => repo.current_view().to_string(), + }; + + let target = match into { + Some(name) => { + require_view(repo, name, "--into")?; + name.to_string() + } + None => match repo.parent_change_count(&source) { + Ok(Some((parent, _))) => parent, + Ok(None) => { + return Err(CliError::InvalidArgument { + message: format!( + "'{source}' is a root view — it has no parent to promote into.\n \ + Pass --into to choose a target (see 'atomic view list')." + ), + }) + } + Err(RepositoryError::ViewNotFound { name }) => { + return Err(unknown_view(repo, &name, "")) + } + Err(e) => return Err(CliError::Repository(e)), + }, + }; + + Ok((source, target)) +} + +/// Fail with an actionable message when a triage view argument names a view +/// that does not exist. +/// +/// The usual cause is typing the placeholder from the help text +/// (`atomic triage review feature`) instead of a real view name, so the message +/// says the argument is optional and names the current view as the value to +/// simply drop in — or the value to leave off entirely. +fn require_view(repo: &Repository, name: &str, arg: &str) -> CliResult<()> { + if repo.view_exists(name).map_err(CliError::Repository)? { + Ok(()) + } else { + Err(unknown_view(repo, name, arg)) + } +} + +/// The error for a view name that does not resolve. +fn unknown_view(repo: &Repository, name: &str, arg: &str) -> CliError { + CliError::InvalidArgument { + message: format!( + "No view named '{name}' — {arg} takes a real view name, not a placeholder.\n \ + Current view is '{}'; omit {arg} to triage that instead, or run \ + 'atomic view list' to see all views.", + repo.current_view() + ), + } +} + +#[cfg(test)] +mod tests { + use clap::{CommandFactory, Parser}; + use tempfile::{tempdir, TempDir}; + + use super::*; + + /// A repo with a root view `dev` and a draft `feature-x` parented on it, + /// checked out on the draft — the state the defaults are meant to serve. + fn draft_repo() -> (Repository, TempDir) { + let dir = tempdir().unwrap(); + let mut repo = Repository::init(dir.path()).unwrap(); + repo.create_draft_view("feature-x", "dev").unwrap(); + repo.set_current_view_in_memory("feature-x"); + (repo, dir) + } + + /// The `` / `--into` pair parsed off a `triage review` argument list. + fn parse_views(args: &[&str]) -> (Option, Option) { + let cmd = + TriageReview::try_parse_from(std::iter::once("review").chain(args.iter().copied())) + .unwrap(); + (cmd.feature, cmd.into) + } + + #[test] + fn both_view_arguments_are_optional() { + assert_eq!(parse_views(&[]), (None, None)); + // A lone source, and a lone target, are each valid on their own. + assert_eq!( + parse_views(&["feature-x"]), + (Some("feature-x".into()), None) + ); + assert_eq!(parse_views(&["--into", "dev"]), (None, Some("dev".into()))); + } + + #[test] + fn usage_advertises_both_arguments_as_optional() { + for cmd in [TriageReview::command(), TriageCandidates::command()] { + // Neither argument may be required, or clap puts it in the usage + // summary instead of `[OPTIONS]` / `[VIEW]`. + for id in ["feature", "into"] { + let arg = cmd + .get_arguments() + .find(|a| a.get_id() == id) + .unwrap_or_else(|| panic!("missing argument `{id}`")); + assert!(!arg.is_required_set(), "`{id}` is still required"); + } + } + + let usage = TriageReview::command().render_usage().to_string(); + assert!(usage.contains("[VIEW]"), "usage: {usage}"); + // The positional is named VIEW, not FEATURE — `feature` read as a + // literal view name, which is exactly the mistake that prompted this. + assert!(!usage.contains("[FEATURE]"), "usage: {usage}"); + } + + #[test] + fn omitted_arguments_resolve_to_current_view_and_its_parent() { + let (repo, _dir) = draft_repo(); + let (feature, into) = resolve_views(&repo, None, None).unwrap(); + assert_eq!(feature, "feature-x"); + assert_eq!(into, "dev"); + } + + #[test] + fn explicit_arguments_pass_through_unchanged() { + let (repo, _dir) = draft_repo(); + let (feature, into) = resolve_views(&repo, Some("dev"), Some("feature-x")).unwrap(); + assert_eq!(feature, "dev"); + assert_eq!(into, "feature-x"); + } + + #[test] + fn omitted_target_follows_the_named_source_not_the_current_view() { + let (repo, _dir) = draft_repo(); + // Current view is feature-x, but naming `dev` as the source must make + // the target *its* parent — there is none, so this is a root-view error. + let err = resolve_views(&repo, Some("dev"), None).unwrap_err(); + assert!(err.to_string().contains("root view"), "unexpected: {err}"); + } + + #[test] + fn a_root_source_with_no_target_explains_the_missing_parent() { + let dir = tempdir().unwrap(); + let repo = Repository::init(dir.path()).unwrap(); + let err = resolve_views(&repo, None, None).unwrap_err(); + let msg = err.to_string(); + assert!(msg.contains("root view"), "unexpected: {msg}"); + assert!(msg.contains("--into"), "unexpected: {msg}"); + } + + #[test] + fn a_placeholder_view_name_is_rejected_with_the_optional_argument_as_the_fix() { + let (repo, _dir) = draft_repo(); + let err = resolve_views(&repo, Some("feature"), None).unwrap_err(); + let msg = err.to_string(); + assert!(msg.contains("No view named 'feature'"), "unexpected: {msg}"); + assert!(msg.contains("omit "), "unexpected: {msg}"); + assert!(msg.contains("feature-x"), "unexpected: {msg}"); + } + + #[test] + fn a_bad_target_is_reported_against_into_not_the_positional() { + let (repo, _dir) = draft_repo(); + let err = resolve_views(&repo, None, Some("nope")).unwrap_err(); + let msg = err.to_string(); + assert!(msg.contains("No view named 'nope'"), "unexpected: {msg}"); + assert!(msg.contains("omit --into"), "unexpected: {msg}"); + } + + #[test] + fn candidates_parses_the_same_optional_pair_as_review() { + let cmd = TriageCandidates::try_parse_from(["candidates"]).unwrap(); + assert!(cmd.feature.is_none()); + assert!(cmd.into.is_none()); + + let cmd = + TriageCandidates::try_parse_from(["candidates", "feature-x", "--into", "dev"]).unwrap(); + assert_eq!(cmd.feature.as_deref(), Some("feature-x")); + assert_eq!(cmd.into.as_deref(), Some("dev")); + } + + #[test] + fn help_carries_examples_for_both_subcommands() { + // The agent help template drops `long_about` but keeps `after_help`, so + // the examples have to live there to be discoverable. + for (name, about) in [ + ("candidates", "atomic triage candidates"), + ("review", "--walkthrough"), + ] { + let help = Triage::command() + .find_subcommand_mut(name) + .unwrap() + .render_long_help() + .to_string(); + assert!( + help.contains(about), + "{name} help missing `{about}`:\n{help}" + ); + assert!( + help.contains("Examples:"), + "{name} help missing examples:\n{help}" + ); + } + } +} diff --git a/atomic-cli/src/main.rs b/atomic-cli/src/main.rs index c1c93ff5..1cd1e275 100644 --- a/atomic-cli/src/main.rs +++ b/atomic-cli/src/main.rs @@ -858,21 +858,11 @@ enum Commands { /// ``` Update(Update), - /// Project the code-review candidate set of a feature view against a target. + /// Project the code-review candidate set of a view against a target. /// - /// Reports the change hashes visible to the feature view but not the - /// target, their transitive dependency-closure additions, and which of - /// those additions are "baggage" (not covered by any intent). - /// - /// # Examples - /// - /// ```text - /// # Human-readable summary - /// atomic triage candidates feature --into dev - /// - /// # Machine-readable worklist - /// atomic triage candidates feature --into dev --json - /// ``` + /// Both arguments are optional: the source view defaults to the current + /// view and the target to that view's parent, so `atomic triage review` + /// with no arguments asks whether the current view is ready to promote. Triage(Triage), /// Generate a shell completion script. From 02d56e056572a3332b1865e291b19e16c149861f Mon Sep 17 00:00:00 2001 From: Aaron Ogle Date: Fri, 25 Sep 2026 19:24:48 -0500 Subject: [PATCH 17/22] feat(change): Ed25519 signing for recorded changes (#214) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit * feat(change): Ed25519 signing for recorded changes Recorded changes previously carried only an unverifiable claim of authorship: the identity's public key was copied into the header, but nothing ever touched the private key. Anyone could author a change claiming any author and any key. This adds real signatures: - New SIGNATURE section (0x04) in the V3 change format carrying the signer's did:atomic fingerprint, an Ed25519 signature over the change's content hash, the signed hash, and a timestamp. - The SIGNATURE section is unhashed, so the change hash remains a pure function of content+header+deps: re-signing with a different key never changes the change's identity. - Signing happens at the Repository::record seam (covers CLI record and revise content-mode); revise --reword signs via Change::sign_with since it bypasses the record pipeline. - Verification is strictly out-of-band: verify_change_signature takes a caller-resolved public key and never treats the DID embedded in the change as a trust root. A forged change (attacker key + claimed identity) fails against the victim's key. - Signature is domain-separated (atomic.change.signature.v1) so a change signature cannot be transplanted from or accepted as a signature over another object kind (attestations, intents, captures). - Backward compatible: unsigned changes load, apply, and display unchanged; a missing signature is "no signature claim", not an error. - When no signing identity is available, record warns and records unsigned instead of silently claiming an author without a key. Tests: 5 integration tests (record->parse->verify byte-level, hash stability, forge scenario, unsigned compat, signed+unsigned coexist) plus unit tests for the signing core. Full workspace suite passes (53 test binaries, 0 failures). * fix(ci): resolve clippy, fmt, and test failures on change-signing branch - revise.rs: replace single-arm `match` with `if let` (clippy::single_match) - change.rs: remove leftover debug eprintln statements from investigation - tests_signing.rs: drop unused glob import, replace debug printlns with assertions, verify the SIGNATURE section round-trips at section level - change_signing_test.rs: remove unnecessary `mut` on repositories that are only read - formatting across all touched files (cargo fmt) Verified locally against the exact CI commands: - cargo fmt --all -- --check: clean - cargo clippy --workspace -- -D warnings: clean - cargo test --workspace: 51 test binaries, 0 failures - CLI semantic diff harness (run_all.sh 09): 1/1 suites passed * feat(agent): sign hook-recorded turns with the effective identity The signing seam covered `atomic record` and `revise` only. Harness turns (opencode, Claude Code, Gemini CLI, ...) go through record_turn, which passed `signing_identity: None`: a turn's header claimed the agent's public key while nothing proved possession of it — exactly the 'unverifiable claim' problem the signing work set out to close. record_turn now resolves a signer through the same levels as the header author, so the signature always proves the header's key claim: 1. A selected delegated agent identity signs with its own key. If the identity resolves but its keypair is not on disk, the turn records unsigned rather than signing as someone else — a signature by any other key would contradict the claim. 2. No selection: plus-tag attribution claims the default identity's key, and the turn signs with it, exactly like `atomic record`. 3. No identities at all: unsigned, legacy behavior unchanged. Attribution and signing share one selected-identity loader, so they can never name different identities. TurnRecordOptions gains an identity_dir override (mirroring AgentAuthorOptions) threading to the author, the signing key, and the envelope's delegation URN — set for testing, None in production. End-to-end tests drive record_turn the way hooks do against a real repo and store: a selected agent identity yields a change whose SIGNATURE section carries the agent's DID and verifies against the agent's public key (and not the human's); no selection signs with the default identity; no store records unsigned. --- Cargo.lock | 4 + atomic-agent/src/identity.rs | 262 ++++++++++++++-- atomic-agent/src/record/mod.rs | 31 +- atomic-agent/src/record/options.rs | 8 + atomic-agent/src/record/provenance.rs | 7 +- atomic-agent/src/record/tests.rs | 201 ++++++++++++ .../src/turn/orchestrator/session_end.rs | 1 + atomic-agent/src/turn/orchestrator/turn.rs | 2 + .../tests/record_view_mismatch_regression.rs | 1 + atomic-cli/src/commands/record/builder.rs | 49 +++ atomic-cli/src/commands/record/command.rs | 18 +- atomic-cli/src/commands/revise.rs | 18 ++ atomic-core/Cargo.toml | 4 + atomic-core/src/change/change.rs | 49 +++ atomic-core/src/change/format_v3/mod.rs | 4 +- .../src/change/format_v3/types/builder.rs | 13 + .../src/change/format_v3/types/header.rs | 17 +- atomic-core/src/change/format_v3/types/mod.rs | 5 +- .../src/change/format_v3/types/section.rs | 162 +++++++++- .../src/change/format_v3/types/tests.rs | 3 +- .../change/format_v3/types/tests_signing.rs | 60 ++++ .../change/format_v3/writer/state_machine.rs | 55 ++++ atomic-core/src/change/mod.rs | 1 + atomic-core/src/change/signing.rs | 224 ++++++++++++++ atomic-core/src/pristine/tables.rs | 10 + atomic-repository/Cargo.toml | 2 + atomic-repository/src/record/mod.rs | 2 +- atomic-repository/src/record/options.rs | 36 +++ .../src/redb_change_store/mod.rs | 17 + .../src/redb_change_store/queries.rs | 24 +- atomic-repository/src/repository/record.rs | 43 ++- .../tests/change_signing_test.rs | 290 ++++++++++++++++++ 32 files changed, 1555 insertions(+), 68 deletions(-) create mode 100644 atomic-core/src/change/format_v3/types/tests_signing.rs create mode 100644 atomic-core/src/change/signing.rs create mode 100644 atomic-repository/tests/change_signing_test.rs diff --git a/Cargo.lock b/Cargo.lock index 830d8ca4..d61e1347 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -238,12 +238,14 @@ dependencies = [ "blake3", "chrono", "data-encoding", + "ed25519-dalek", "fastcdc", "log", "parking_lot", "postcard", "quickcheck", "quickcheck_macros", + "rand 0.8.7", "rayon", "redb 2.6.3", "redb 4.2.0", @@ -334,7 +336,9 @@ dependencies = [ "atomic-identity", "atomic-semantic", "chrono", + "data-encoding", "dirs", + "ed25519-dalek", "flate2", "fs2", "ignore", diff --git a/atomic-agent/src/identity.rs b/atomic-agent/src/identity.rs index 34831dd0..cfbbb2d4 100644 --- a/atomic-agent/src/identity.rs +++ b/atomic-agent/src/identity.rs @@ -143,19 +143,7 @@ pub fn active_delegation_urn( agent_identity: Option<&str>, identity_dir: Option<&Path>, ) -> Option { - let name = agent_identity - .map(str::to_string) - .or_else(|| std::env::var(AGENT_IDENTITY_ENV).ok()) - .map(|n| n.trim().to_string()) - .filter(|n| !n.is_empty())?; - - let store = match identity_dir { - Some(dir) => atomic_identity::IdentityStore::open(dir), - None => atomic_identity::IdentityStore::open_default(), - } - .ok()?; - - let identity = store.load_by_name(&name).ok()?; + let (store, identity) = load_selected_agent_identity(agent_identity, identity_dir)?; atomic_canonical::delegation::active_for_delegate(&store, &identity) .map(|d| d.delegation.id.to_urn()) } @@ -166,26 +154,37 @@ pub fn active_delegation_urn( /// gets both authenticated pushes and correctly attributed changes. pub const AGENT_IDENTITY_ENV: &str = "ATOMIC_AGENT_IDENTITY"; -/// Build an author from a delegated agent identity, if one is configured and -/// resolvable. +/// Open the identity store at the given directory, or the default one. /// -/// Returns `None` — rather than failing — whenever the identity is missing or -/// unreadable. Recording a turn must not break because an agent identity was -/// mistyped; falling back to the plus-tag author keeps the work attributed to -/// *someone* and leaves a debug log explaining why it is not keyed. -fn delegated_agent_author(options: &AgentAuthorOptions<'_>) -> Option { - let name = options - .agent_identity - .clone() +/// Returns `None` rather than failing: every caller here is on a +/// best-effort path where recording must continue without identity data. +fn open_identity_store(identity_dir: Option<&Path>) -> Option { + match identity_dir { + Some(dir) => atomic_identity::IdentityStore::open(dir).ok(), + None => atomic_identity::IdentityStore::open_default().ok(), + } +} + +/// The currently-selected delegated agent identity, if one resolves. +/// +/// Shared by the author path ([`delegated_agent_author`]) and the signing +/// path ([`resolve_turn_signer`]) so attribution and proof can never name +/// different identities: same name chain (explicit option, else +/// [`AGENT_IDENTITY_ENV`]), same store, same refusal of human identities. +/// +/// Returns the store alongside the identity — callers need it to load the +/// keypair without re-opening (and possibly disagreeing about) the store. +fn load_selected_agent_identity( + agent_identity: Option<&str>, + identity_dir: Option<&Path>, +) -> Option<(atomic_identity::IdentityStore, atomic_identity::Identity)> { + let name = agent_identity + .map(str::to_string) .or_else(|| std::env::var(AGENT_IDENTITY_ENV).ok()) .map(|n| n.trim().to_string()) .filter(|n| !n.is_empty())?; - let store = match options.identity_dir.as_deref() { - Some(dir) => atomic_identity::IdentityStore::open(dir), - None => atomic_identity::IdentityStore::open_default(), - } - .ok()?; + let store = open_identity_store(identity_dir)?; let identity = match store.load_by_name(&name) { Ok(identity) => identity, @@ -203,6 +202,79 @@ fn delegated_agent_author(options: &AgentAuthorOptions<'_>) -> Option { return None; } + Some((store, identity)) +} + +/// Resolve the signer for a recorded turn: the identity whose public key +/// the header author claims. +/// +/// The levels mirror [`resolve_agent_author`] exactly, so the signature +/// always proves the header's key claim: +/// +/// 1. A **selected delegated agent identity** — keyed attribution claims +/// its public key, so the turn signs with its secret key. If the +/// identity resolves but its keypair is not on disk (a +/// verification-only identity), the turn records **unsigned** rather +/// than silently signing as someone else — a signature by a different +/// key would contradict the header's claim. +/// 2. The **default identity** — the plus-tag path claims the human's +/// public key, so the turn signs with the human's secret key, exactly +/// like `atomic record` does. +/// 3. Neither — `None`; the change records unsigned (legacy behavior). +/// +/// Never fails: a turn must not fail to record over identity selection. +pub fn resolve_turn_signer( + agent_identity: Option<&str>, + identity_dir: Option<&Path>, +) -> Option { + use atomic_canonical::did::did_for_public_key; + use atomic_identity::{Identity, KeyPair}; + + fn signing_identity( + identity: &Identity, + keypair: &KeyPair, + ) -> atomic_repository::record::SigningIdentity { + atomic_repository::record::SigningIdentity { + signer_did: did_for_public_key(&identity.public_key), + secret_key: *keypair.secret.as_bytes(), + } + } + + // Level 1: the selected delegated identity — claim and proof must agree. + if let Some((store, identity)) = load_selected_agent_identity(agent_identity, identity_dir) { + return match store.load_keypair(&identity.id, None) { + Ok(keypair) => Some(signing_identity(&identity, &keypair)), + Err(e) => { + log::debug!( + "Agent identity '{}' has no usable local keypair ({e}); \ + recording unsigned rather than signing as someone else", + identity.name + ); + None + } + }; + } + + // Level 2: plus-tag attribution claims the default identity's key. + let store = open_identity_store(identity_dir)?; + let identity = store.get_default().ok()??; + let keypair = store.load_keypair(&identity.id, None).ok()?; + Some(signing_identity(&identity, &keypair)) +} + +/// Build an author from a delegated agent identity, if one is configured and +/// resolvable. +/// +/// Returns `None` — rather than failing — whenever the identity is missing or +/// unreadable. Recording a turn must not break because an agent identity was +/// mistyped; falling back to the plus-tag author keeps the work attributed to +/// *someone* and leaves a debug log explaining why it is not keyed. +fn delegated_agent_author(options: &AgentAuthorOptions<'_>) -> Option { + let (_, identity) = load_selected_agent_identity( + options.agent_identity.as_deref(), + options.identity_dir.as_deref(), + )?; + let session_short = extract_session_short(options.session_id); let tag = format!( "{}+{}", @@ -832,6 +904,140 @@ mod tests { ); } + // resolve_turn_signer + + /// Test fixture: a store holding a human default identity (with key) + /// and a delegated agent identity (with key). + fn signer_test_store( + dir: &Path, + ) -> ( + atomic_identity::KeyPair, + atomic_identity::Identity, + atomic_identity::KeyPair, + atomic_identity::Identity, + ) { + use atomic_identity::{Identity, IdentityStore, IdentityType, KeyPair}; + + let mut store = IdentityStore::open(dir).unwrap(); + + let human_key = KeyPair::generate(); + let human = Identity::builder("alice") + .email("alice@example.com") + .public_key(human_key.public.clone()) + .build() + .unwrap(); + store.save_with_keypair(&human, &human_key, None).unwrap(); + store.set_default(&human.id).unwrap(); + + let agent_key = KeyPair::generate(); + let agent = Identity::builder("alice+claude") + .identity_type(IdentityType::Agent) + .email("alice+claude@example.com") + .public_key(agent_key.public.clone()) + .delegated_by(human.id) + .build() + .unwrap(); + store.save_with_keypair(&agent, &agent_key, None).unwrap(); + + (human_key, human, agent_key, agent) + } + + /// The whole point of the feature: a selected agent identity signs with + /// its OWN key, so the signature proves the header's key claim. The + /// human's key must not sign work attributed to the agent. + #[test] + fn a_selected_agent_identity_signs_with_its_own_key() { + use atomic_canonical::did::did_for_public_key; + + let dir = TempDir::new().unwrap(); + let (human_key, _, agent_key, agent) = signer_test_store(dir.path()); + + let signer = resolve_turn_signer(Some("alice+claude"), Some(dir.path())).expect("signer"); + + assert_eq!( + signer.signer_did, + did_for_public_key(&agent.public_key), + "signer DID must name the agent identity" + ); + assert_eq!(signer.secret_key, *agent_key.secret.as_bytes()); + assert_ne!( + signer.secret_key, + *human_key.secret.as_bytes(), + "the human's key must never sign agent-attributed work" + ); + } + + /// A verification-only agent identity (no keypair on disk) must not fall + /// through to signing as someone else: the header claims the agent's + /// key, so a signature by any other key would contradict the claim. + #[test] + fn a_keyless_selected_identity_records_unsigned_not_as_someone_else() { + use atomic_identity::{Identity, IdentityStore, IdentityType}; + + let dir = TempDir::new().unwrap(); + let mut store = IdentityStore::open(dir.path()).unwrap(); + + let human_key = atomic_identity::KeyPair::generate(); + let human = Identity::builder("alice") + .email("alice@example.com") + .public_key(human_key.public.clone()) + .build() + .unwrap(); + store.save_with_keypair(&human, &human_key, None).unwrap(); + store.set_default(&human.id).unwrap(); + + let agent_key = atomic_identity::KeyPair::generate(); + let agent = Identity::builder("alice+claude") + .identity_type(IdentityType::Agent) + .public_key(agent_key.public.clone()) + .delegated_by(human.id) + .build() + .unwrap(); + // Saved WITHOUT the keypair — the store knows the identity and its + // public key, but not the secret. + store.save(&agent).unwrap(); + + assert!(resolve_turn_signer(Some("alice+claude"), Some(dir.path())).is_none()); + } + + /// Plus-tag attribution claims the default identity's public key, so + /// that is what signs when no agent identity is selected — the same + /// identity `atomic record` would sign with. + #[test] + fn with_no_selection_the_default_identity_signs() { + use atomic_canonical::did::did_for_public_key; + + let dir = TempDir::new().unwrap(); + let (human_key, human, _, _) = signer_test_store(dir.path()); + + let signer = resolve_turn_signer(None, Some(dir.path())).expect("signer"); + + assert_eq!(signer.signer_did, did_for_public_key(&human.public_key)); + assert_eq!(signer.secret_key, *human_key.secret.as_bytes()); + } + + /// A human identity passed as the agent identity is refused at level 1 — + /// attribution falls to the plus-tag path, so the signer must be the + /// default identity, matching the header's claim. + #[test] + fn a_human_identity_selected_falls_back_to_the_default_signer() { + let dir = TempDir::new().unwrap(); + let (human_key, _, _, _) = signer_test_store(dir.path()); + + // "alice" is the human default, not an agent identity. + let signer = resolve_turn_signer(Some("alice"), Some(dir.path())).expect("signer"); + assert_eq!(signer.secret_key, *human_key.secret.as_bytes()); + } + + /// No identities at all: no signer, and the turn records unsigned — + /// legacy behavior, unchanged. + #[test] + fn an_empty_store_has_no_signer() { + let dir = TempDir::new().unwrap(); + std::fs::create_dir_all(dir.path().join("identities")).unwrap(); + assert!(resolve_turn_signer(None, Some(dir.path().join("identities").as_path())).is_none()); + } + // resolve_agent_author (integration) #[test] diff --git a/atomic-agent/src/record/mod.rs b/atomic-agent/src/record/mod.rs index df46caaa..ac5d48dc 100644 --- a/atomic-agent/src/record/mod.rs +++ b/atomic-agent/src/record/mod.rs @@ -71,7 +71,6 @@ use atomic_core::types::Base32; use atomic_repository::status::RepositoryStatus; use crate::error::{AgentError, AgentResult}; -use crate::identity::build_agent_author; use crate::transcript; // Re-export primary types @@ -99,12 +98,13 @@ fn build_turn_header( ) -> ChangeHeader { let message = build_turn_message(options, status, untracked_paths); - let author = build_agent_author( - &options.session.agent_name, - &options.session.agent_display_name, - &options.session.session_id, - options.agent_identity.as_deref(), - ); + let author = crate::identity::resolve_agent_author(&crate::identity::AgentAuthorOptions { + agent_name: &options.session.agent_name, + agent_display_name: &options.session.agent_display_name, + session_id: &options.session.session_id, + identity_dir: options.identity_dir.clone(), + agent_identity: options.agent_identity.clone(), + }); ChangeHeader::builder() .message(message) @@ -423,6 +423,23 @@ pub fn record_turn( record_options = record_options.with_all(false).paths(status_files.clone()); } + // Sign the turn with the identity whose public key the header claims: + // the selected agent identity when one is in force, else the default + // identity behind the plus-tag. `None` records unsigned (legacy + // behavior) — and never fails the turn. + match crate::identity::resolve_turn_signer( + options.agent_identity.as_deref(), + options.identity_dir.as_deref(), + ) { + Some(signing) => { + log::debug!("Signing turn as {}", signing.signer_did); + record_options = record_options.with_signing_identity(signing); + } + None => { + log::debug!("Recording turn unsigned: no signing identity resolved"); + } + } + let mut outcome = match repo.record(header, record_options) { Ok(outcome) => outcome, Err(atomic_repository::record::RecordError::NothingToRecord) => { diff --git a/atomic-agent/src/record/options.rs b/atomic-agent/src/record/options.rs index f4a6b11d..4c2db2d4 100644 --- a/atomic-agent/src/record/options.rs +++ b/atomic-agent/src/record/options.rs @@ -48,6 +48,14 @@ pub struct TurnRecordOptions<'a> { /// author falls back to the plus-tag of the default identity, exactly /// as before agent identities existed. pub agent_identity: Option, + + /// Override for the identity store directory. + /// + /// If `None`, uses `~/.atomic/identities/`. Set this for testing — it + /// threads to every identity resolution the turn records (author, + /// signing key, delegation URN on the envelope) so they all see the + /// same store. + pub identity_dir: Option, } /// The result of recording a turn as an Atomic change. diff --git a/atomic-agent/src/record/provenance.rs b/atomic-agent/src/record/provenance.rs index c1107e2c..695e7558 100644 --- a/atomic-agent/src/record/provenance.rs +++ b/atomic-agent/src/record/provenance.rs @@ -244,9 +244,10 @@ pub(crate) fn build_turn_envelope( // has no delegated identity — the plus-tag path signs with the human's key // and there is no certificate to point at, which is precisely the // difference the field is there to record. - if let Some(urn) = - crate::identity::active_delegation_urn(options.agent_identity.as_deref(), None) - { + if let Some(urn) = crate::identity::active_delegation_urn( + options.agent_identity.as_deref(), + options.identity_dir.as_deref(), + ) { builder = builder.delegation_id(urn); } diff --git a/atomic-agent/src/record/tests.rs b/atomic-agent/src/record/tests.rs index 3ed699bd..86161d9c 100644 --- a/atomic-agent/src/record/tests.rs +++ b/atomic-agent/src/record/tests.rs @@ -36,6 +36,7 @@ fn make_options<'a>(session: &'a AgentSession, event: &'a TurnEvent) -> TurnReco turn_duration_ms: 12400, prompt: Some("Fix the authentication bug in login.rs".to_string()), agent_identity: None, + identity_dir: None, } } @@ -871,6 +872,7 @@ fn test_record_turn_nonexistent_repo_fails() { turn_duration_ms: 5000, prompt: Some("Fix the bug".to_string()), agent_identity: None, + identity_dir: None, }; let result = record_turn(Path::new("/nonexistent/repo/path"), &options); @@ -1112,6 +1114,7 @@ fn test_orphaned_session_view_duplicates_content_on_merge() { turn_duration_ms: 1000, prompt: Some("Bump step10".to_string()), agent_identity: None, + identity_dir: None, }; record_turn(repo_root, &options_a).unwrap(); @@ -1138,6 +1141,7 @@ fn test_orphaned_session_view_duplicates_content_on_merge() { turn_duration_ms: 1000, prompt: Some("Bump step70".to_string()), agent_identity: None, + identity_dir: None, }; record_turn(repo_root, &options_b) .expect("record_turn should self-heal an orphaned session view rather than fail"); @@ -1316,3 +1320,200 @@ fn scoped_snapshot_includes_requested_files_restored_to_clean_and_deletions() { assert!(after["files"].as_object().unwrap().contains_key("a.txt")); assert!(after["files"]["a.txt"].is_null()); } + +// Turn signing (end-to-end) +// +// A turn recorded with a selected agent identity must carry a signature by +// the AGENT's key — the signature proves the header author's key claim, and +// the two must name the same identity. These tests drive `record_turn` the +// way hooks do, against a real repository and a real identity store. + +/// Set up a repository with one recorded file and a forked session view, +/// ready for a turn to be recorded into. +fn signed_turn_repo(repo_root: &Path) -> String { + use atomic_repository::Repository; + + std::fs::write(repo_root.join("sign-me.txt"), "first\n").unwrap(); + { + let mut repo = Repository::init(repo_root).unwrap(); + repo.add( + "sign-me.txt", + atomic_repository::tracking::TrackingOptions::default(), + ) + .unwrap(); + let header = atomic_core::change::ChangeHeader::new("Add sign-me.txt"); + let options = atomic_repository::record::RecordOptions::new() + .with_all(true) + .save_to_store(true) + .apply_after_record(true); + repo.record(header, options).unwrap(); + repo.create_view_from("session-signed", "dev").unwrap(); + } + + "first\n".to_string() +} + +fn record_signed_turn( + repo_root: &Path, + initial: &str, + agent_identity: Option<&str>, + identity_dir: Option, +) -> crate::record::TurnRecordOutcome { + let mut session = AgentSession::new("signed-sess", "claude-code", "Claude Code"); + session.view_name = "session-signed".to_string(); + session.set_parent_view("dev"); + + std::fs::write(repo_root.join("sign-me.txt"), "second\n").unwrap(); + debug_assert_ne!(initial, "second\n"); + + let event = TurnEvent::new("signed-sess", HookType::TurnEnd); + let options = TurnRecordOptions { + session: &session, + event: &event, + turn_number: 1, + turn_duration_ms: 1000, + prompt: Some("Edit sign-me.txt".to_string()), + agent_identity: agent_identity.map(str::to_string), + identity_dir, + }; + record_turn(repo_root, &options).unwrap() +} + +#[test] +fn a_turn_with_a_selected_agent_identity_is_signed_with_the_agents_key() { + use atomic_canonical::did::did_for_public_key; + use atomic_core::change::signing::verify_change_signature; + use atomic_identity::{Identity, IdentityStore, IdentityType, KeyPair}; + use atomic_repository::Repository; + use tempfile::TempDir; + + // An identity store holding a human default and a delegated agent + // identity, each with its own keypair on disk. + let id_dir = TempDir::new().unwrap(); + let mut store = IdentityStore::open(id_dir.path()).unwrap(); + let human_key = KeyPair::generate(); + let human = Identity::builder("alice") + .email("alice@example.com") + .public_key(human_key.public.clone()) + .build() + .unwrap(); + store.save_with_keypair(&human, &human_key, None).unwrap(); + store.set_default(&human.id).unwrap(); + let agent_key = KeyPair::generate(); + let agent = Identity::builder("alice+claude") + .identity_type(IdentityType::Agent) + .email("alice+claude@example.com") + .public_key(agent_key.public.clone()) + .delegated_by(human.id) + .build() + .unwrap(); + store.save_with_keypair(&agent, &agent_key, None).unwrap(); + drop(store); + + let repo_dir = TempDir::new().unwrap(); + let initial = signed_turn_repo(repo_dir.path()); + let outcome = record_signed_turn( + repo_dir.path(), + &initial, + Some("alice+claude"), + Some(id_dir.path().to_path_buf()), + ); + + // The turn's change carries a signature, made with the agent's key. + let repo = Repository::open_existing(repo_dir.path()).unwrap(); + let change = repo.load_change(&outcome.hash).unwrap(); + let signature = change.signature.as_ref().expect("turn must be signed"); + assert_eq!( + signature.signer_did, + did_for_public_key(&agent.public_key), + "the signer must be the agent identity" + ); + verify_change_signature(signature, agent.public_key.as_bytes(), &outcome.hash) + .expect("signature verifies against the agent's public key"); + + // Attribution and proof name the same identity: the header claims the + // agent's public key and the signature is made with its secret. + let author = change + .hashed + .header + .authors + .first() + .expect("turn change has an author"); + assert_eq!( + author.identity.as_deref(), + Some(agent.public_key_base32().as_str()) + ); + assert_ne!( + author.identity.as_deref(), + Some(human.public_key_base32().as_str()) + ); + + // And the human's key must NOT verify it — this is the forge direction. + assert!( + verify_change_signature(signature, human.public_key.as_bytes(), &outcome.hash).is_err() + ); +} + +#[test] +fn a_turn_without_a_selected_identity_signs_with_the_default_identity() { + use atomic_canonical::did::did_for_public_key; + use atomic_core::change::signing::verify_change_signature; + use atomic_identity::{Identity, IdentityStore, KeyPair}; + use atomic_repository::Repository; + use tempfile::TempDir; + + // No agent identity selected — plus-tag attribution claims the human's + // key, and the turn signs with it, exactly like `atomic record`. + let id_dir = TempDir::new().unwrap(); + let mut store = IdentityStore::open(id_dir.path()).unwrap(); + let human_key = KeyPair::generate(); + let human = Identity::builder("alice") + .email("alice@example.com") + .public_key(human_key.public.clone()) + .build() + .unwrap(); + store.save_with_keypair(&human, &human_key, None).unwrap(); + store.set_default(&human.id).unwrap(); + drop(store); + + let repo_dir = TempDir::new().unwrap(); + let initial = signed_turn_repo(repo_dir.path()); + let outcome = record_signed_turn( + repo_dir.path(), + &initial, + None, + Some(id_dir.path().to_path_buf()), + ); + + let repo = Repository::open_existing(repo_dir.path()).unwrap(); + let change = repo.load_change(&outcome.hash).unwrap(); + let signature = change.signature.as_ref().expect("turn must be signed"); + assert_eq!(signature.signer_did, did_for_public_key(&human.public_key)); + verify_change_signature(signature, human.public_key.as_bytes(), &outcome.hash) + .expect("signature verifies against the human's public key"); +} + +#[test] +fn a_turn_with_no_identity_store_records_unsigned() { + use atomic_repository::Repository; + use tempfile::TempDir; + + // No identities anywhere: recording still works, and the change carries + // no signature — legacy behavior, unchanged. + let id_dir = TempDir::new().unwrap(); + let repo_dir = TempDir::new().unwrap(); + let initial = signed_turn_repo(repo_dir.path()); + let outcome = record_signed_turn( + repo_dir.path(), + &initial, + None, + Some(id_dir.path().to_path_buf()), + ); + + let repo = Repository::open_existing(repo_dir.path()).unwrap(); + let change = repo.load_change(&outcome.hash).unwrap(); + assert!( + change.signature.is_none(), + "no identity store means no signature claim" + ); +} diff --git a/atomic-agent/src/turn/orchestrator/session_end.rs b/atomic-agent/src/turn/orchestrator/session_end.rs index 6b2adb34..9750ca8a 100644 --- a/atomic-agent/src/turn/orchestrator/session_end.rs +++ b/atomic-agent/src/turn/orchestrator/session_end.rs @@ -118,6 +118,7 @@ impl TurnOrchestrator { turn_duration_ms, prompt, agent_identity: self.agent_identity.clone(), + identity_dir: None, }; record_turn(&self.repo_root, &record_options) }; diff --git a/atomic-agent/src/turn/orchestrator/turn.rs b/atomic-agent/src/turn/orchestrator/turn.rs index d05b5156..24ea1dda 100644 --- a/atomic-agent/src/turn/orchestrator/turn.rs +++ b/atomic-agent/src/turn/orchestrator/turn.rs @@ -205,6 +205,7 @@ impl TurnOrchestrator { turn_duration_ms: 0, prompt: None, agent_identity: self.agent_identity.clone(), + identity_dir: None, }; if crate::record::scope::has_pending_changes(&self.repo_root, &options)? { return Err(AgentError::RecordFailed { @@ -330,6 +331,7 @@ impl TurnOrchestrator { turn_duration_ms, prompt, agent_identity: self.agent_identity.clone(), + identity_dir: None, }; match record_turn(&self.repo_root, &record_options) { diff --git a/atomic-agent/tests/record_view_mismatch_regression.rs b/atomic-agent/tests/record_view_mismatch_regression.rs index 6eaca5d1..81496f71 100644 --- a/atomic-agent/tests/record_view_mismatch_regression.rs +++ b/atomic-agent/tests/record_view_mismatch_regression.rs @@ -31,6 +31,7 @@ fn options<'a>( turn_duration_ms: 1000, prompt: Some(prompt.to_string()), agent_identity: None, + identity_dir: None, } } diff --git a/atomic-cli/src/commands/record/builder.rs b/atomic-cli/src/commands/record/builder.rs index d197eb1b..372315a4 100644 --- a/atomic-cli/src/commands/record/builder.rs +++ b/atomic-cli/src/commands/record/builder.rs @@ -112,6 +112,55 @@ impl Record { Ok(None) } + /// Resolve the signing identity for this change. + /// + /// Mirrors `resolve_author`'s precedence: `--identity` flag, then + /// `--usage` default, then the global default identity. Loads the + /// identity's keypair from the store. Returns `None` when no identity is + /// available (the change records unsigned, with a warning). + pub(super) fn resolve_signing_identity( + &self, + ) -> CliResult> { + use atomic_canonical::did::did_for_public_key; + + let store = match IdentityStore::open_default() { + Ok(store) => store, + Err(_) => return Ok(None), + }; + + // 1. --identity flag + let identity = if let Some(identity_name) = &self.identity { + store + .load_by_name(identity_name) + .map_err(|_| CliError::IdentityNotFound(identity_name.clone()))? + // 2. --usage default + } else if let Some(usage_str) = &self.usage { + let usage = IdentityUsage::parse(usage_str); + match store.get_default_for_usage(&usage) { + Ok(Some(identity)) => identity, + _ => return Ok(None), + } + // 3. Global default + } else { + match store.get_default() { + Ok(Some(identity)) => identity, + _ => return Ok(None), + } + }; + + // Load the private key. A verification-only identity (no secret key + // on disk) cannot sign — record unsigned rather than fail. + let keypair = match store.load_keypair(&identity.id, None) { + Ok(keypair) => keypair, + Err(_) => return Ok(None), + }; + + Ok(Some(atomic_repository::record::SigningIdentity { + signer_did: did_for_public_key(&identity.public_key), + secret_key: *keypair.secret.as_bytes(), + })) + } + /// Get the commit message, potentially from editor. pub(super) fn get_message(&self) -> CliResult { // If message was provided, use it diff --git a/atomic-cli/src/commands/record/command.rs b/atomic-cli/src/commands/record/command.rs index 912f525f..081eef68 100644 --- a/atomic-cli/src/commands/record/command.rs +++ b/atomic-cli/src/commands/record/command.rs @@ -40,7 +40,23 @@ impl Command for Record { let header = header_builder.build(); // Build record options - let options = self.build_options()?; + let mut options = self.build_options()?; + + // Resolve the signing identity and attach signing credentials. + // The signing identity is the same one that supplied the author. + // When no identity is available the change records unsigned. + match self.resolve_signing_identity()? { + Some(signing) => { + options = options.with_signing_identity(signing); + } + None => { + print_warning( + "No signing identity available — recording change UNSIGNED. \ + Others cannot verify authorship. Run `atomic identity new` \ + and `atomic identity default ` to enable signing.", + ); + } + } // If --all, first add all untracked files if self.all { diff --git a/atomic-cli/src/commands/revise.rs b/atomic-cli/src/commands/revise.rs index 9a6ed3fd..2e2a62f0 100644 --- a/atomic-cli/src/commands/revise.rs +++ b/atomic-cli/src/commands/revise.rs @@ -78,6 +78,7 @@ use std::path::PathBuf; use atomic_core::change::{Change, ChangeHeader}; use atomic_core::types::{Base32, Hash}; +use atomic_identity::IdentityStore; use atomic_repository::{ HistoryEntry, HistoryOptions, RecordOptions, Repository, StatusOptions, UnrecordOptions, }; @@ -687,6 +688,23 @@ impl Revise { original_change.hashed.dependencies.clone(), ); + // Sign the reworded change with the current identity. Reword bypasses + // the record pipeline (it rebuilds the change directly), so it must + // sign here — a revised change is signed like any new record. + let mut new_change = new_change; + if let Ok(store) = IdentityStore::open_default() { + let signing_identity = store.get_default().ok().flatten(); + if let Some(identity) = signing_identity { + if let Ok(keypair) = store.load_keypair(&identity.id, None) { + let _ = new_change.sign_with( + &atomic_canonical::did::did_for_public_key(&identity.public_key), + keypair.secret.as_bytes(), + chrono::Utc::now().timestamp(), + ); + } + } + } + // Save the new change let new_hash = repo.save_change(&new_change).map_err(|e| { CliError::Internal(anyhow::anyhow!("Failed to save reworded change: {}", e)) diff --git a/atomic-core/Cargo.toml b/atomic-core/Cargo.toml index d0f43dbd..74f7c294 100644 --- a/atomic-core/Cargo.toml +++ b/atomic-core/Cargo.toml @@ -24,6 +24,10 @@ fastcdc.workspace = true blake3.workspace = true xxhash-rust.workspace = true +# Signing (change signatures) +ed25519-dalek.workspace = true +rand.workspace = true + # Compression zstd.workspace = true diff --git a/atomic-core/src/change/change.rs b/atomic-core/src/change/change.rs index d4973f36..61c62133 100644 --- a/atomic-core/src/change/change.rs +++ b/atomic-core/src/change/change.rs @@ -125,6 +125,13 @@ pub struct Change { /// Useful for storing editor metadata, review comments, AI transcripts, etc. pub unhashed: Option, + /// Optional Ed25519 signature over the change's content hash. + /// + /// Excluded from the change hash (the SIGNATURE section is unhashed), so + /// re-signing with a different key never changes the change's identity. + /// `None` for unsigned (legacy) changes. + pub signature: Option, + /// Binary content blob /// /// This contains the actual file content referenced by hunks. @@ -180,6 +187,7 @@ impl Change { contents_hash, }, unhashed: None, + signature: None, contents, } } @@ -200,6 +208,7 @@ impl Change { contents_hash: Hash::of(&[]), }, unhashed: None, + signature: None, contents: Vec::new(), } } @@ -326,6 +335,7 @@ impl Change { // 2. Compute section counts let has_provenance = !self.hashed.provenance.is_empty(); let has_unhashed = self.unhashed.is_some(); + let has_signature = self.signature.is_some(); let has_content = !self.contents.is_empty(); // Chunk content with FastCDC for delta transfer + parallel compression @@ -361,6 +371,9 @@ impl Change { if has_unhashed { file_header_builder = file_header_builder.with_unhashed(); } + if has_signature { + file_header_builder = file_header_builder.with_signature(); + } let file_header = file_header_builder.build(); @@ -386,6 +399,12 @@ impl Change { change_writer.write_provenance(&self.hashed.provenance)?; } + // Write signature if present (unhashed section — does not affect the + // change hash; must be written while still in WRITING_METADATA state) + if let Some(ref signature) = self.signature { + change_writer.write_signature(signature)?; + } + // 6. Write GRAPH section(s) — compact graph ops if !self.hashed.hunks.is_empty() { let compactor = format_v3::compact::Compactor::new(&hash_table); @@ -501,6 +520,7 @@ impl Change { let mut file_ops: Vec = Vec::new(); let mut contents: Vec = Vec::new(); let mut unhashed: Option = None; + let mut signature: Option = None; while let Some(section) = change_reader.next_section()? { let section_type = section.section_type; @@ -559,6 +579,13 @@ impl Change { SectionType::Unhashed => { unhashed = Some(serde_json::from_slice(§ion.payload)?); } + SectionType::Signature => { + signature = Some(postcard::from_bytes(§ion.payload).map_err(|error| { + ChangeError::Invalid(format!( + "failed to deserialize {section_type} section: {error}" + )) + })?); + } } } @@ -586,6 +613,7 @@ impl Change { contents_hash, }, unhashed, + signature, contents, }; @@ -597,6 +625,27 @@ impl Change { self.hashed.dependencies.contains(hash) } + /// Sign this change in place with an Ed25519 secret key. + /// + /// Computes the change's content hash (unaffected by signatures — the + /// SIGNATURE section is unhashed), signs it, and attaches the signature. + /// The change's identity does not change. + /// + /// Used by record and revise paths that bypass the assemble/serialize + /// pipeline (e.g. `revise --reword`). + pub fn sign_with( + &mut self, + signer_did: &str, + secret_key_bytes: &[u8; 32], + timestamp: i64, + ) -> Result { + let hash = self.hash()?; + let signature = + crate::change::signing::sign_change(signer_did, secret_key_bytes, &hash, timestamp); + self.signature = Some(signature); + Ok(hash) + } + /// Check if this change knows about another change. /// /// A change "knows" another if it's either a dependency or extra_known. diff --git a/atomic-core/src/change/format_v3/mod.rs b/atomic-core/src/change/format_v3/mod.rs index 45337af3..c71568e5 100644 --- a/atomic-core/src/change/format_v3/mod.rs +++ b/atomic-core/src/change/format_v3/mod.rs @@ -202,8 +202,8 @@ pub use error::{FormatError, FormatResult, FORMAT_VERSION, MAGIC, MAX_HASH_TABLE // Core types pub use types::{ - CompactPosition, ContentChunkHeader, FileHeader, FileHeaderBuilder, FileHeaderFlags, - SectionHeader, SectionType, Trailer, + ChangeSignature, CompactPosition, ContentChunkHeader, FileHeader, FileHeaderBuilder, + FileHeaderFlags, SectionHeader, SectionType, Trailer, }; // Hash index types and helpers diff --git a/atomic-core/src/change/format_v3/types/builder.rs b/atomic-core/src/change/format_v3/types/builder.rs index 37536409..e5af74da 100644 --- a/atomic-core/src/change/format_v3/types/builder.rs +++ b/atomic-core/src/change/format_v3/types/builder.rs @@ -42,6 +42,7 @@ pub struct FileHeaderBuilder { total_uncompressed: u64, has_provenance: bool, has_unhashed: bool, + has_signature: bool, } impl FileHeaderBuilder { @@ -55,6 +56,7 @@ impl FileHeaderBuilder { total_uncompressed: 0, has_provenance: false, has_unhashed: false, + has_signature: false, } } @@ -106,6 +108,14 @@ impl FileHeaderBuilder { self } + /// Mark that this change has a signature section. + /// + /// Automatically sets the `HAS_SIGNATURE` flag. + pub fn with_signature(mut self) -> Self { + self.has_signature = true; + self + } + /// Build the [`FileHeader`], auto-computing flags. pub fn build(self) -> FileHeader { let mut flags = FileHeaderFlags::NONE; @@ -118,6 +128,9 @@ impl FileHeaderBuilder { if self.has_unhashed { flags.set(FileHeaderFlags::HAS_UNHASHED); } + if self.has_signature { + flags.set(FileHeaderFlags::HAS_SIGNATURE); + } FileHeader { magic: MAGIC, diff --git a/atomic-core/src/change/format_v3/types/header.rs b/atomic-core/src/change/format_v3/types/header.rs index 8ea6202d..074f09a3 100644 --- a/atomic-core/src/change/format_v3/types/header.rs +++ b/atomic-core/src/change/format_v3/types/header.rs @@ -27,11 +27,12 @@ use std::io::{Read, Write}; /// | 0 | `HAS_PROVENANCE` | Provenance section is present | /// | 1 | `HAS_SEMANTIC` | Semantic sections are present | /// | 2 | `HAS_UNHASHED` | Unhashed section is present | -/// | 3-31 | Reserved | Must be zero | +/// | 3 | `HAS_SIGNATURE` | Signature section is present | +/// | 4-31 | Reserved | Must be zero | /// /// # Forward Compatibility /// -/// Readers MUST ignore unknown flags (bits 3-31). This allows newer writers +/// Readers MUST ignore unknown flags (bits 4-31). This allows newer writers /// to set flags that older readers don't understand without breaking them. /// If a future flag requires breaking changes, the `version` field should /// be incremented instead. @@ -51,8 +52,12 @@ impl FileHeaderFlags { /// Unhashed section is present in the file. pub const HAS_UNHASHED: u32 = 1 << 2; + /// Signature section is present in the file. + pub const HAS_SIGNATURE: u32 = 1 << 3; + /// Mask of all known flags (for validation). - const KNOWN_MASK: u32 = Self::HAS_PROVENANCE | Self::HAS_SEMANTIC | Self::HAS_UNHASHED; + const KNOWN_MASK: u32 = + Self::HAS_PROVENANCE | Self::HAS_SEMANTIC | Self::HAS_UNHASHED | Self::HAS_SIGNATURE; /// Create flags from a raw `u32` value. /// @@ -113,6 +118,9 @@ impl fmt::Display for FileHeaderFlags { if self.has(Self::HAS_UNHASHED) { parts.push("UNHASHED"); } + if self.has(Self::HAS_SIGNATURE) { + parts.push("SIGNATURE"); + } if parts.is_empty() { write!(f, "(none)") } else { @@ -368,6 +376,9 @@ impl FileHeader { if self.flags.has(FileHeaderFlags::HAS_UNHASHED) { count += 1; } + if self.flags.has(FileHeaderFlags::HAS_SIGNATURE) { + count += 1; + } count } diff --git a/atomic-core/src/change/format_v3/types/mod.rs b/atomic-core/src/change/format_v3/types/mod.rs index d7b0c43d..d35bcf6c 100644 --- a/atomic-core/src/change/format_v3/types/mod.rs +++ b/atomic-core/src/change/format_v3/types/mod.rs @@ -74,6 +74,9 @@ pub mod section; #[cfg(test)] mod tests; +#[cfg(test)] +mod tests_signing; + // ── Re-exports ───────────────────────────────────────────────────────── // Hash index types @@ -86,4 +89,4 @@ pub use header::{FileHeader, FileHeaderFlags}; pub use builder::{FileHeaderBuilder, Trailer}; // Section types -pub use section::{ContentChunkHeader, SectionHeader, SectionType}; +pub use section::{ChangeSignature, ContentChunkHeader, SectionHeader, SectionType}; diff --git a/atomic-core/src/change/format_v3/types/section.rs b/atomic-core/src/change/format_v3/types/section.rs index 63dee330..5e1ab214 100644 --- a/atomic-core/src/change/format_v3/types/section.rs +++ b/atomic-core/src/change/format_v3/types/section.rs @@ -20,7 +20,7 @@ use std::io::{Read, Write}; /// /// | Range | Layer | Types | /// |-------|-------|-------| -/// | `0x01-0x0F` | Metadata | HEADER, DEPS, PROVENANCE | +/// | `0x01-0x0F` | Metadata | HEADER, DEPS, PROVENANCE, SIGNATURE | /// | `0x10-0x1F` | Graph | GRAPH (one per file) | /// | `0x20-0x2F` | Content | CONTENT (content-defined chunks) | /// | `0x30-0x3F` | Semantic | SEMANTIC (one per file) | @@ -33,10 +33,11 @@ use std::io::{Read, Write}; /// 1. `HEADER` (exactly one) /// 2. `DEPS` (exactly one, may be empty) /// 3. `PROVENANCE` (zero or one) -/// 4. `GRAPH` sections (zero or more, one per file) -/// 5. `SEMANTIC` sections (zero or more, one per file) -/// 6. `CONTENT` chunks (zero or more) -/// 7. `UNHASHED` (zero or one) +/// 4. `SIGNATURE` (zero or one) +/// 5. `GRAPH` sections (zero or more, one per file) +/// 6. `SEMANTIC` sections (zero or more, one per file) +/// 7. `CONTENT` chunks (zero or more) +/// 8. `UNHASHED` (zero or one) /// /// This ordering ensures that: /// - The graph layer can be read without seeking past semantic/content sections @@ -60,6 +61,18 @@ pub enum SectionType { /// Payload: `zstd(postcard(Vec))` Provenance = 0x03, + /// Cryptographic signature over the change's hashed sections (optional). + /// + /// Payload: `zstd(postcard(ChangeSignature))` — the signer's DID, the + /// Ed25519 signature over the content hash, and a timestamp. + /// + /// SIGNATURE is NOT hashed (like UNHASHED): the change hash must remain + /// a pure function of content+header+deps so re-signing the same change + /// with a different key never changes its identity. The signature covers + /// the file's content hash (the trailer hash), which is computed over + /// everything except UNHASHED and SIGNATURE sections. + Signature = 0x04, + /// Graph operations for a single file (storage/merge layer). /// /// There is one GRAPH section per modified file. Each section contains @@ -127,6 +140,7 @@ impl SectionType { 0x01 => Ok(SectionType::Header), 0x02 => Ok(SectionType::Dependencies), 0x03 => Ok(SectionType::Provenance), + 0x04 => Ok(SectionType::Signature), 0x10 => Ok(SectionType::Graph), 0x20 => Ok(SectionType::Content), 0x30 => Ok(SectionType::Semantic), @@ -153,8 +167,10 @@ impl SectionType { /// Returns `true` if this section type is part of the hashed content. /// - /// All sections except `Unhashed` contribute to the change's content hash. - /// The hash is computed incrementally as hashed sections are written. + /// All sections except `Unhashed` and `Signature` contribute to the + /// change's content hash. The hash is computed incrementally as hashed + /// sections are written. SIGNATURE is excluded so the change hash stays + /// a pure function of content+header+deps (re-signing never re-hashes). /// /// # Examples /// @@ -166,13 +182,14 @@ impl SectionType { /// assert!(SectionType::Semantic.is_hashed()); /// assert!(SectionType::Content.is_hashed()); /// assert!(!SectionType::Unhashed.is_hashed()); + /// assert!(!SectionType::Signature.is_hashed()); /// ``` #[inline] pub const fn is_hashed(self) -> bool { - !matches!(self, SectionType::Unhashed) + !matches!(self, SectionType::Unhashed | SectionType::Signature) } - /// Returns `true` if this is a metadata section (HEADER, DEPS, PROVENANCE). + /// Returns `true` if this is a metadata section (HEADER, DEPS, PROVENANCE, SIGNATURE). /// /// Metadata sections appear first in the file and contain information /// about the change itself rather than file modifications. @@ -180,7 +197,10 @@ impl SectionType { pub const fn is_metadata(self) -> bool { matches!( self, - SectionType::Header | SectionType::Dependencies | SectionType::Provenance + SectionType::Header + | SectionType::Dependencies + | SectionType::Provenance + | SectionType::Signature ) } @@ -210,6 +230,7 @@ impl SectionType { SectionType::Header => "HEADER", SectionType::Dependencies => "DEPS", SectionType::Provenance => "PROVENANCE", + SectionType::Signature => "SIGNATURE", SectionType::Graph => "GRAPH", SectionType::Content => "CONTENT", SectionType::Semantic => "SEMANTIC", @@ -229,10 +250,11 @@ impl SectionType { SectionType::Header => 0, SectionType::Dependencies => 1, SectionType::Provenance => 2, - SectionType::Graph => 3, - SectionType::Semantic => 4, - SectionType::Content => 5, - SectionType::Unhashed => 6, + SectionType::Signature => 3, + SectionType::Graph => 4, + SectionType::Semantic => 5, + SectionType::Content => 6, + SectionType::Unhashed => 7, } } } @@ -443,3 +465,115 @@ impl ContentChunkHeader { self.compressed_len as f64 / self.uncompressed_len as f64 } } + +// ═══════════════════════════════════════════════════════════════════════ +// ChangeSignature — payload of the SIGNATURE section +// ═══════════════════════════════════════════════════════════════════════ + +/// Cryptographic signature over a change's content hash (V3). +/// +/// The signature binds the signer's identity to the exact set of hashed +/// sections that made up the change at record time. It is an Ed25519 +/// signature over the **content hash** (the trailer hash) — the blake3 hash +/// covering the hash table and all hashed sections. The SIGNATURE section +/// itself is excluded from that hash (like UNHASHED), so: +/// +/// - the change hash is a pure function of content+header+deps — re-signing +/// with a different key never changes the change's identity; +/// - verification needs no re-serialization: the signed hash is stored in +/// the section and can be compared against the trailer hash directly. +/// +/// Verifying a signature means: check `signed_hash` equals the file's +/// verified content hash, then check the Ed25519 signature over +/// `signed_hash` against a *caller-supplied* public key. The DID embedded +/// here is a discovery hint only — never the trust root. +/// +/// # Fields +/// +/// - `signer_did`: `did:atomic:` fingerprint of the +/// signer's key. Non-reversible — verification needs a key resolved +/// out-of-band (identity store, `identity lookup-key`, caller argument). +/// - `signature`: Ed25519 signature (64 bytes) over the unsigned content hash. +/// - `signed_hash`: the unsigned content hash that was signed (kept here so +/// verification can detect which level was signed without re-serializing). +/// - `timestamp`: when the signature was made (Unix epoch seconds). +/// - `method`: signature scheme identifier, currently always `ed25519`. +#[derive(Clone, Debug, PartialEq, Eq, Serialize, Deserialize)] +pub struct ChangeSignature { + /// DID of the signer (`did:atomic:...`). A discovery hint, not a trust + /// root — verification must resolve the key out-of-band. + pub signer_did: String, + + /// Ed25519 signature (64 bytes) over `signed_hash`. + #[serde(with = "serde_bytes_array_64")] + pub signature: [u8; 64], + + /// The unsigned content hash that was signed. + #[serde(with = "serde_bytes_array_32")] + pub signed_hash: [u8; 32], + + /// When the signature was made (Unix epoch seconds). + pub timestamp: i64, + + /// Signature scheme identifier. Always `ed25519` for this version. + pub method: String, +} + +impl ChangeSignature { + /// The signature method identifier for Ed25519. + pub const METHOD_ED25519: &'static str = "ed25519"; + + /// Size of an Ed25519 signature in bytes. + pub const SIGNATURE_SIZE: usize = 64; + + /// Create a new change signature. + pub fn new( + signer_did: impl Into, + signature: [u8; 64], + signed_hash: [u8; 32], + timestamp: i64, + ) -> Self { + Self { + signer_did: signer_did.into(), + signature, + signed_hash, + timestamp, + method: Self::METHOD_ED25519.to_string(), + } + } + + /// Returns `true` if this signature uses the Ed25519 method. + pub fn is_ed25519(&self) -> bool { + self.method == Self::METHOD_ED25519 + } +} + +/// serde helper: fixed-size 64-byte array as a plain byte vector. +mod serde_bytes_array_64 { + use serde::{Deserialize, Deserializer, Serialize, Serializer}; + + pub fn serialize(v: &[u8; 64], s: S) -> Result { + v.as_slice().serialize(s) + } + + pub fn deserialize<'de, D: Deserializer<'de>>(d: D) -> Result<[u8; 64], D::Error> { + let vec = Vec::::deserialize(d)?; + vec.try_into() + .map_err(|_| serde::de::Error::custom("expected 64-byte Ed25519 signature")) + } +} + +/// serde helper: fixed-size 32-byte array as a plain byte vector. +mod serde_bytes_array_32 { + use serde::{Deserialize, Deserializer, Serialize, Serializer}; + + pub fn serialize(v: &[u8; 32], s: S) -> Result { + v.as_slice().serialize(s) + } + + pub fn deserialize<'de, D: Deserializer<'de>>(d: D) -> Result<[u8; 32], D::Error> { + let vec = Vec::::deserialize(d)?; + vec.try_into() + .map_err(|_| serde::de::Error::custom("expected 32-byte hash")) + } +} diff --git a/atomic-core/src/change/format_v3/types/tests.rs b/atomic-core/src/change/format_v3/types/tests.rs index 1bc0523a..b070bef9 100644 --- a/atomic-core/src/change/format_v3/types/tests.rs +++ b/atomic-core/src/change/format_v3/types/tests.rs @@ -219,7 +219,7 @@ fn test_section_type_from_byte_roundtrip() { #[test] fn test_section_type_from_byte_invalid() { // Test some invalid bytes - for byte in [0x00, 0x04, 0x0F, 0x11, 0x21, 0x31, 0xFF] { + for byte in [0x00, 0x0F, 0x11, 0x21, 0x31, 0xFF] { let result = SectionType::from_byte(byte); assert!(result.is_err(), "byte 0x{:02X} should be invalid", byte); } @@ -230,6 +230,7 @@ fn test_section_type_is_hashed() { assert!(SectionType::Header.is_hashed()); assert!(SectionType::Dependencies.is_hashed()); assert!(SectionType::Provenance.is_hashed()); + assert!(!SectionType::Signature.is_hashed()); assert!(SectionType::Graph.is_hashed()); assert!(SectionType::Content.is_hashed()); assert!(SectionType::Semantic.is_hashed()); diff --git a/atomic-core/src/change/format_v3/types/tests_signing.rs b/atomic-core/src/change/format_v3/types/tests_signing.rs new file mode 100644 index 00000000..2cc79b15 --- /dev/null +++ b/atomic-core/src/change/format_v3/types/tests_signing.rs @@ -0,0 +1,60 @@ +//! Roundtrip checks for the ChangeSignature section. +#[cfg(test)] +mod tests { + use super::super::section::ChangeSignature; + + #[test] + fn change_signature_postcard_roundtrip() { + let sig = ChangeSignature::new("did:atomic:test", [7u8; 64], [9u8; 32], 42); + let bytes = postcard::to_allocvec(&sig).expect("serialize"); + let decoded: ChangeSignature = postcard::from_bytes(&bytes).expect("deserialize"); + assert_eq!(sig, decoded); + } +} + +#[cfg(test)] +mod full_change_roundtrip { + use crate::change::format_v3::{ChangeReader, SectionType}; + use crate::change::{signing, Author, Change, ChangeHeader}; + use crate::types::Hash; + + #[test] + fn signed_change_serializes_and_deserializes() { + let header = ChangeHeader::builder() + .message("test") + .author(Author::new("T", Some("t@t.dev"))) + .build(); + let mut change = Change::new(header, Vec::new(), b"hello world\n".to_vec(), Vec::new()); + let hash = change + .sign_with("did:atomic:test", &[3u8; 32], 12345) + .expect("sign"); + assert!(change.signature.is_some()); + + let mut bytes = Vec::new(); + let re_hash = change.serialize(&mut bytes).expect("serialize"); + assert_eq!(re_hash, hash, "signature must not change hash"); + + // The SIGNATURE section survives a section-level round-trip. + let mut saw_signature = false; + { + let mut cursor = std::io::Cursor::new(&bytes); + let mut reader = ChangeReader::open(&mut cursor).expect("open"); + while let Some(section) = reader.next_section().expect("section") { + if section.section_type == SectionType::Signature { + let decoded_sig: crate::change::format_v3::ChangeSignature = + postcard::from_bytes(§ion.payload).expect("standalone decode"); + assert_eq!(decoded_sig.timestamp, 12345); + saw_signature = true; + } + } + } + assert!(saw_signature, "SIGNATURE section present in file"); + let (decoded, decoded_hash) = + Change::deserialize(&mut std::io::Cursor::new(&bytes)).expect("deserialize"); + assert_eq!(decoded_hash, hash); + assert!(decoded.signature.is_some()); + assert_eq!(decoded.signature, change.signature); + let _ = Hash::of(b""); + let _ = signing::CHANGE_SIGNATURE_DOMAIN; + } +} diff --git a/atomic-core/src/change/format_v3/writer/state_machine.rs b/atomic-core/src/change/format_v3/writer/state_machine.rs index dc4dc3c2..c0d5bb34 100644 --- a/atomic-core/src/change/format_v3/writer/state_machine.rs +++ b/atomic-core/src/change/format_v3/writer/state_machine.rs @@ -154,6 +154,9 @@ pub struct ChangeWriter<'w, W: Write> { /// Tracks whether the DEPS section has been written. pub(super) wrote_deps_section: bool, + /// Tracks whether the SIGNATURE section has been written. + pub(super) wrote_signature_section: bool, + /// The highest section ordering value written so far. /// Used to enforce monotonic section ordering. pub(super) last_section_ordering: Option, @@ -187,6 +190,7 @@ impl<'w, W: Write> ChangeWriter<'w, W> { stats: WriterStats::default(), wrote_header_section: false, wrote_deps_section: false, + wrote_signature_section: false, last_section_ordering: None, } } @@ -389,6 +393,57 @@ impl<'w, W: Write> ChangeWriter<'w, W> { Ok(()) } + /// Write the optional SIGNATURE section. + /// + /// Contains an Ed25519 signature over the change's *unsigned* content + /// hash, made with the signer's private key. The SIGNATURE section is a + /// hashed section, so it must be written with the writer's hasher paused + /// relative to the signature itself: the signature covers the unsigned + /// hash (computed by serializing without this section), while the file's + /// trailer hash covers the file including this section. See + /// [`ChangeSignature`](crate::change::format_v3::ChangeSignature) for the + /// two-level hashing model. + /// + /// # Arguments + /// + /// * `signature` - The signature payload (signer DID, Ed25519 signature, + /// signed hash, timestamp, method). + /// + /// # Errors + /// + /// - [`FormatError::UnexpectedSection`] if DEPS hasn't been written yet, + /// or if SIGNATURE was already written. + /// - Postcard serialization errors. + /// - Zstd compression errors. + /// - I/O errors. + pub fn write_signature( + &mut self, + signature: &crate::change::format_v3::ChangeSignature, + ) -> FormatResult<()> { + if self.state != WriterState::WritingMetadata { + return Err(FormatError::UnexpectedSection { + got: "SIGNATURE".to_string(), + expected: format!("state WRITING_METADATA, but was {}", self.state.name()), + }); + } + if !self.wrote_deps_section { + return Err(FormatError::UnexpectedSection { + got: "SIGNATURE".to_string(), + expected: "DEPS section must be written first".to_string(), + }); + } + if self.wrote_signature_section { + return Err(FormatError::UnexpectedSection { + got: "SIGNATURE (duplicate)".to_string(), + expected: "SIGNATURE section already written".to_string(), + }); + } + + self.write_postcard_section(SectionType::Signature, signature)?; + self.wrote_signature_section = true; + Ok(()) + } + // ── Steps 4-7 (graph, semantic, content, unhashed) are in sections.rs ── // ── Step 8: Finalize ─────────────────────────────────────────── diff --git a/atomic-core/src/change/mod.rs b/atomic-core/src/change/mod.rs index 75be0d9c..c016fe5a 100644 --- a/atomic-core/src/change/mod.rs +++ b/atomic-core/src/change/mod.rs @@ -102,6 +102,7 @@ pub mod ops; mod provenance; pub mod provenance_graph; pub mod session; +pub mod signing; mod store; // Re-export all public types diff --git a/atomic-core/src/change/signing.rs b/atomic-core/src/change/signing.rs new file mode 100644 index 00000000..6ffd8e3b --- /dev/null +++ b/atomic-core/src/change/signing.rs @@ -0,0 +1,224 @@ +//! Change signing — Ed25519 signatures over a change's content hash. +//! +//! This module provides the cryptographic core of change signing: +//! - [`sign_change`]: sign a change's content hash with a secret key +//! - [`verify_change_signature`]: verify a signature with a *caller-supplied* +//! public key +//! +//! # Trust Model (critical) +//! +//! The public key used for verification MUST come from outside the change +//! being verified — an identity store lookup, `identity lookup-key`, or a +//! caller argument. The `signer_did` embedded in the signature is a +//! *discovery hint*, never a trust root: an attacker can sign garbage with +//! their own keypair and embed their own DID. Self-referential verification +//! (checking the signature against the key carried in the change) validates +//! nothing and is deliberately not provided here. + +use crate::change::format_v3::ChangeSignature; +use crate::types::Hash; + +/// Domain separator for change signatures. Ed25519 is deterministic, so the +/// domain keeps a change signature from being transplanted into (or accepted +/// from) a signature made over a different object kind (attestation, intent, +/// capture) with the same key. +pub const CHANGE_SIGNATURE_DOMAIN: &[u8] = b"atomic.change.signature.v1"; + +/// Sign a change's content hash with an Ed25519 secret key. +/// +/// `secret_key_bytes` is the 32-byte Ed25519 seed +/// (`atomic_identity::SecretKey::as_bytes`). `signer_did` is the signer's +/// `did:atomic:...` fingerprint (see `atomic_canonical::did`). +/// +/// The signature is made over `domain || 0 || content_hash`, so a signature +/// over a raw hash (or another object's signature input) never verifies as a +/// change signature. +pub fn sign_change( + signer_did: &str, + secret_key_bytes: &[u8; 32], + content_hash: &Hash, + timestamp: i64, +) -> ChangeSignature { + use ed25519_dalek::Signer; + + let signing_key = ed25519_dalek::SigningKey::from_bytes(secret_key_bytes); + let message = signature_message(content_hash); + let signature = signing_key.sign(&message).to_bytes(); + + ChangeSignature::new(signer_did, signature, *content_hash.as_bytes(), timestamp) +} + +/// Verify a change signature against a *caller-supplied* Ed25519 public key. +/// +/// The caller resolves the key out-of-band (identity store, lookup-key +/// remote, explicit argument). This function deliberately takes no key from +/// the signature itself — self-referential verification is the broken trust +/// model this module exists to avoid. +/// +/// # Arguments +/// +/// * `signature` — the signature section from the change. +/// * `public_key_bytes` — the verifier's *known* Ed25519 public key for the +/// claimed signer. +/// * `content_hash` — the change's verified content hash (e.g. from the +/// trailer after `Change::deserialize`). +/// +/// # Errors +/// +/// Returns an error if the signature was not made over this change's hash +/// with the matching private key, or if the signature section is malformed +/// (wrong method, hash mismatch with the change's actual hash). +pub fn verify_change_signature( + signature: &ChangeSignature, + public_key_bytes: &[u8; 32], + content_hash: &Hash, +) -> Result<(), ChangeSignatureError> { + use ed25519_dalek::Verifier; + + if !signature.is_ed25519() { + return Err(ChangeSignatureError::UnsupportedMethod { + method: signature.method.clone(), + }); + } + + // The signed hash must be THIS change's hash — a signature over some + // other change's hash is meaningless even if it verifies cryptographically. + if signature.signed_hash != *content_hash.as_bytes() { + return Err(ChangeSignatureError::HashMismatch { + signed: baseline32(&signature.signed_hash), + actual: crate::types::Base32::to_base32(content_hash), + }); + } + + let verifying_key = ed25519_dalek::VerifyingKey::from_bytes(public_key_bytes) + .map_err(|_| ChangeSignatureError::InvalidPublicKey)?; + + let sig = ed25519_dalek::Signature::from_bytes(&signature.signature); + verifying_key + .verify(&signature_message(content_hash), &sig) + .map_err(|_| ChangeSignatureError::SignatureVerificationFailed { + signer: signature.signer_did.clone(), + }) +} + +/// The canonical signed message: domain separator, then the content hash. +fn signature_message(content_hash: &Hash) -> Vec { + let mut message = Vec::with_capacity(CHANGE_SIGNATURE_DOMAIN.len() + 1 + 32); + message.extend_from_slice(CHANGE_SIGNATURE_DOMAIN); + message.push(0); + message.extend_from_slice(content_hash.as_bytes()); + message +} + +/// Base32-encode raw bytes for error messages. +fn baseline32(bytes: &[u8; 32]) -> String { + data_encoding::BASE32_NOPAD.encode(bytes) +} + +/// Errors from change signature verification. +#[derive(Debug, thiserror::Error)] +pub enum ChangeSignatureError { + /// The signature section uses a method this version does not verify. + #[error("unsupported signature method: `{method}`")] + UnsupportedMethod { + /// The unsupported method identifier. + method: String, + }, + + /// The signature was made over a different change's hash. + #[error("signature was made over hash {signed}, but this change's hash is {actual}")] + HashMismatch { + /// The hash the signature actually covers. + signed: String, + /// This change's actual content hash. + actual: String, + }, + + /// The caller-supplied public key is not a valid Ed25519 key. + #[error("invalid public key")] + InvalidPublicKey, + + /// The Ed25519 signature does not verify against the supplied key. + #[error("signature verification failed for signer `{signer}` (key was resolved out-of-band)")] + SignatureVerificationFailed { + /// The DID claimed by the signature. + signer: String, + }, +} + +#[cfg(test)] +mod tests { + use super::*; + use ed25519_dalek::{Signer as _, SigningKey}; + use rand::RngCore; + + fn test_keypair() -> (SigningKey, [u8; 32]) { + let mut seed = [0u8; 32]; + rand::rngs::OsRng.fill_bytes(&mut seed); + let signing_key = SigningKey::from_bytes(&seed); + (signing_key, seed) + } + + #[test] + fn sign_and_verify_roundtrip() { + let (signing_key, seed) = test_keypair(); + let hash = Hash::of(b"test change content"); + let did = "did:atomic:TESTSIGNER"; + + let sig = sign_change(did, &seed, &hash, 1739290034); + assert!(sig.is_ed25519()); + assert_eq!(sig.signer_did, did); + assert_eq!(sig.signed_hash, *hash.as_bytes()); + + let public = signing_key.verifying_key().to_bytes(); + verify_change_signature(&sig, &public, &hash).unwrap(); + } + + #[test] + fn tampered_hash_fails() { + let (signing_key, seed) = test_keypair(); + let hash = Hash::of(b"original"); + let sig = sign_change("did:atomic:x", &seed, &hash, 0); + + let other = Hash::of(b"tampered"); + let public = signing_key.verifying_key().to_bytes(); + let err = verify_change_signature(&sig, &public, &other).unwrap_err(); + assert!(matches!(err, ChangeSignatureError::HashMismatch { .. })); + } + + #[test] + fn wrong_key_fails() { + let (_signing_key, seed) = test_keypair(); + let (_other_key, other_seed) = test_keypair(); + let hash = Hash::of(b"content"); + let sig = sign_change("did:atomic:attacker", &seed, &hash, 0); + + // Verify against a DIFFERENT (victim's) key — the forge scenario. + let victim_key = SigningKey::from_bytes(&other_seed) + .verifying_key() + .to_bytes(); + + let err = verify_change_signature(&sig, &victim_key, &hash).unwrap_err(); + assert!(matches!( + err, + ChangeSignatureError::SignatureVerificationFailed { .. } + )); + } + + #[test] + fn domain_separation() { + // A signature made over the raw hash must NOT verify as a change + // signature (domain separation). + let (signing_key, _seed) = test_keypair(); + let hash = Hash::of(b"content"); + let raw_sig = signing_key.sign(hash.as_bytes()).to_bytes(); + + let sig = ChangeSignature::new("did:atomic:x", raw_sig, *hash.as_bytes(), 0); + let public = signing_key.verifying_key().to_bytes(); + let err = verify_change_signature(&sig, &public, &hash).unwrap_err(); + assert!(matches!( + err, + ChangeSignatureError::SignatureVerificationFailed { .. } + )); + } +} diff --git a/atomic-core/src/pristine/tables.rs b/atomic-core/src/pristine/tables.rs index 269bcaa8..80715c7d 100644 --- a/atomic-core/src/pristine/tables.rs +++ b/atomic-core/src/pristine/tables.rs @@ -505,6 +505,16 @@ pub const CHANGE_CHUNKS: TableDefinition<&[u8; 36], &[u8; 32]> = pub const CHANGE_UNHASHED: TableDefinition<&[u8; 32], &[u8]> = TableDefinition::new("change_unhashed"); +/// Change signatures (Ed25519 over the content hash). +/// +/// Key: change content hash (blake3, 32 bytes) +/// Value: compressed postcard `ChangeSignature` +/// +/// Like CHANGE_UNHASHED, the signature does not affect the change's identity — +/// re-signing the same change with a different key never changes its hash. +pub const CHANGE_SIGNATURES: TableDefinition<&[u8; 32], &[u8]> = + TableDefinition::new("change_signatures"); + // Pending provenance journal tables used by the redb-native change store. pub const PROVENANCE_STORE_META: TableDefinition<&str, u64> = TableDefinition::new("provenance_store_meta"); diff --git a/atomic-repository/Cargo.toml b/atomic-repository/Cargo.toml index 397db592..47bd8d18 100644 --- a/atomic-repository/Cargo.toml +++ b/atomic-repository/Cargo.toml @@ -43,3 +43,5 @@ fs2 = "0.4" [dev-dependencies] tempfile = { workspace = true } +ed25519-dalek = { workspace = true } +data-encoding = { workspace = true } diff --git a/atomic-repository/src/record/mod.rs b/atomic-repository/src/record/mod.rs index ccc656d6..0168f74d 100644 --- a/atomic-repository/src/record/mod.rs +++ b/atomic-repository/src/record/mod.rs @@ -69,7 +69,7 @@ mod assemble; mod options; pub use assemble::{build_header, filter_files, RecordOutcome, RecordStats}; -pub use options::RecordOptions; +pub use options::{RecordOptions, SigningIdentity}; use atomic_core::record::workflow::{AssemblyError, GlobalizeError}; use thiserror::Error; diff --git a/atomic-repository/src/record/options.rs b/atomic-repository/src/record/options.rs index 7e6343eb..6f9ebeb3 100644 --- a/atomic-repository/src/record/options.rs +++ b/atomic-repository/src/record/options.rs @@ -104,6 +104,25 @@ pub struct RecordOptions { /// Defaults to `true`. Internal callers that already know every path they /// intend to record may disable this to avoid an unrelated untracked scan. detect_raw_renames: bool, + + /// Signing credentials: who signs this change, and with what key. + /// + /// When set, the recorded change carries a SIGNATURE section — an + /// Ed25519 signature over the change's content hash made with the + /// supplied secret key. When `None`, the change is recorded unsigned + /// (legacy behavior). + signing_identity: Option, +} + +/// Credentials for signing a recorded change. +#[derive(Debug, Clone)] +pub struct SigningIdentity { + /// The signer's DID (`did:atomic:...`). A discovery hint only — the + /// trust root is always the caller-resolved public key. + pub signer_did: String, + + /// The signer's Ed25519 seed (32 bytes). + pub secret_key: [u8; 32], } impl RecordOptions { @@ -471,6 +490,23 @@ impl Default for RecordOptions { provenance: Vec::new(), allow_conflict_markers: false, detect_raw_renames: true, + signing_identity: None, } } } + +impl RecordOptions { + /// Set the signing identity for this change. + /// + /// When set, the recorded change carries an Ed25519 signature over its + /// content hash. See [`SigningIdentity`]. + pub fn with_signing_identity(mut self, signing: SigningIdentity) -> Self { + self.signing_identity = Some(signing); + self + } + + /// The signing identity, if configured. + pub fn signing_identity(&self) -> Option<&SigningIdentity> { + self.signing_identity.as_ref() + } +} diff --git a/atomic-repository/src/redb_change_store/mod.rs b/atomic-repository/src/redb_change_store/mod.rs index 4f9c4f23..1dd6e65f 100644 --- a/atomic-repository/src/redb_change_store/mod.rs +++ b/atomic-repository/src/redb_change_store/mod.rs @@ -203,6 +203,9 @@ pub struct StoredChangeMeta { /// Whether unhashed data is stored. pub has_unhashed: bool, + + /// Whether a signature is stored. + pub has_signature: bool, } // ═══════════════════════════════════════════════════════════════════════ @@ -259,6 +262,7 @@ impl RedbChangeStore { let _ = txn.open_table(tables::CONTENT_CHUNKS)?; let _ = txn.open_table(tables::CHANGE_CHUNKS)?; let _ = txn.open_table(tables::CHANGE_UNHASHED)?; + let _ = txn.open_table(tables::CHANGE_SIGNATURES)?; Self::initialize_provenance_tables(&txn)?; } txn.commit()?; @@ -317,6 +321,7 @@ impl RedbChangeStore { let mut semantic_sections: Vec<(u32, Vec)> = Vec::new(); let mut content_chunks: Vec<(u32, [u8; 32], Vec)> = Vec::new(); let mut unhashed_payload: Option> = None; + let mut signature_payload: Option> = None; let mut graph_idx = 0u32; let mut semantic_idx = 0u32; @@ -356,6 +361,9 @@ impl RedbChangeStore { SectionType::Unhashed => { unhashed_payload = Some(section.payload.clone()); } + SectionType::Signature => { + signature_payload = Some(section.payload.clone()); + } } } @@ -375,6 +383,7 @@ impl RedbChangeStore { content_chunk_count: content_chunks.len() as u32, has_provenance: provenance_payload.is_some(), has_unhashed: unhashed_payload.is_some(), + has_signature: signature_payload.is_some(), }; // Serialize and compress metadata @@ -432,6 +441,14 @@ impl RedbChangeStore { .map_err(|e| RedbStoreError::Serialization(e.to_string()))?; unhashed_table.insert(&content_hash, compressed.as_slice())?; } + + // CHANGE_SIGNATURES + if let Some(signature) = &signature_payload { + let mut sig_table = txn.open_table(tables::CHANGE_SIGNATURES)?; + let compressed = zstd::encode_all(signature.as_slice(), 3) + .map_err(|e| RedbStoreError::Serialization(e.to_string()))?; + sig_table.insert(&content_hash, compressed.as_slice())?; + } } txn.commit()?; diff --git a/atomic-repository/src/redb_change_store/queries.rs b/atomic-repository/src/redb_change_store/queries.rs index 2527a8ea..1f5c13d9 100644 --- a/atomic-repository/src/redb_change_store/queries.rs +++ b/atomic-repository/src/redb_change_store/queries.rs @@ -1,7 +1,7 @@ //! Query, statistics, and export operations for the redb change store. use atomic_core::change::format_v3::{ - ChangeWriter, FileHeader, HashDedupTable, SectionType, WriterOptions, + ChangeSignature, ChangeWriter, FileHeader, HashDedupTable, SectionType, WriterOptions, }; use atomic_core::change::Change; use atomic_core::pristine::tables; @@ -400,6 +400,9 @@ impl RedbChangeStore { if meta.has_unhashed { file_header_builder = file_header_builder.with_unhashed(); } + if meta.has_signature { + file_header_builder = file_header_builder.with_signature(); + } let file_header = file_header_builder.build(); @@ -416,8 +419,25 @@ impl RedbChangeStore { // Write DEPS section writer.write_dependencies(&meta.dependency_indices)?; - // Write GRAPH sections + // Write SIGNATURE section (metadata phase — before GRAPH). The + // payload is the stored postcard ChangeSignature; decompress from + // CHANGE_SIGNATURES and write as the SIGNATURE section. let txn = self.db().begin_read()?; + { + let sig_table = txn.open_table(tables::CHANGE_SIGNATURES)?; + if let Some(value) = sig_table.get(hash)? { + let compressed = value.value(); + let sig_bytes = zstd::decode_all(compressed).map_err(|e| { + RedbStoreError::Corrupt(format!("signature decompression failed: {}", e)) + })?; + let signature: ChangeSignature = postcard::from_bytes(&sig_bytes).map_err(|e| { + RedbStoreError::Corrupt(format!("signature deserialization failed: {}", e)) + })?; + writer.write_signature(&signature)?; + } + } + + // Write GRAPH sections { let graph_table = txn.open_table(tables::CHANGE_GRAPH)?; for idx in 0..meta.graph_section_count { diff --git a/atomic-repository/src/repository/record.rs b/atomic-repository/src/repository/record.rs index ffbda3f3..31050715 100644 --- a/atomic-repository/src/repository/record.rs +++ b/atomic-repository/src/repository/record.rs @@ -1009,15 +1009,48 @@ impl Repository { // Serialize to V3 format and compute content hash. // We keep the raw V3 bytes so we can save them directly to disk // without re-serializing (which would produce a different hash). + // + // Signing: the SIGNATURE section is unhashed, so the content hash is + // a pure function of the unsigned content. We therefore serialize + // unsigned first, obtain the content hash, sign it, then re-serialize + // with the signature attached — the hash is identical both times and + // the change's identity is unaffected by who signed it. let mut v3_bytes = Vec::new(); let computed_hash = change .serialize(&mut v3_bytes) .map_err(|e| RecordError::ChangeStore(e.to_string()))?; - // Reload the change from the V3 buffer to get a clean deserialized form - let (final_change, verified_hash) = Change::deserialize(&mut v3_bytes.as_slice()) - .map_err(|e| RecordError::ChangeStore(e.to_string()))?; - debug_assert_eq!(computed_hash, verified_hash); + let change = if let Some(signing) = options.signing_identity() { + let signature = atomic_core::change::signing::sign_change( + &signing.signer_did, + &signing.secret_key, + &computed_hash, + chrono::Utc::now().timestamp(), + ); + let mut signed = change; + signed.signature = Some(signature); + let mut signed_bytes = Vec::new(); + let signed_hash = signed + .serialize(&mut signed_bytes) + .map_err(|e| RecordError::ChangeStore(e.to_string()))?; + debug_assert_eq!( + signed_hash, computed_hash, + "SIGNATURE section must not affect the change hash" + ); + // Round-trip the signed bytes for a clean deserialized form + let (final_signed, verified_hash) = + Change::deserialize(&mut signed_bytes.as_slice()) + .map_err(|e| RecordError::ChangeStore(e.to_string()))?; + debug_assert_eq!(signed_hash, verified_hash); + v3_bytes = signed_bytes; + final_signed + } else { + // Unsigned path: round-trip the unsigned bytes as before + let (final_change, verified_hash) = Change::deserialize(&mut v3_bytes.as_slice()) + .map_err(|e| RecordError::ChangeStore(e.to_string()))?; + debug_assert_eq!(computed_hash, verified_hash); + final_change + }; if trace_record { eprintln!( @@ -1026,7 +1059,7 @@ impl Repository { ); } - let mut outcome = RecordOutcome::new(final_change, computed_hash, stats); + let mut outcome = RecordOutcome::new(change, computed_hash, stats); // Stash the original V3 bytes so save_change can write them directly // instead of re-serializing (which may produce a different hash). outcome.set_v3_bytes(v3_bytes); diff --git a/atomic-repository/tests/change_signing_test.rs b/atomic-repository/tests/change_signing_test.rs new file mode 100644 index 00000000..1150bbea --- /dev/null +++ b/atomic-repository/tests/change_signing_test.rs @@ -0,0 +1,290 @@ +//! Integration tests for change signing (CB: intent ATOM::aaron::27). +//! +//! Covers: +//! - AC-1: record with a signing identity produces a `.change` file whose +//! signature verifies against the identity's public key; the SIGNATURE +//! section does not affect the change hash. +//! - AC-3: out-of-band verification — the forge scenario (attacker signs +//! with their own key and embeds their own DID) must fail verification +//! against the victim's known key. +//! - AC-5: backward compatibility — unsigned changes still load and apply; +//! the change hash is identical with and without a signature. + +use std::fs; +use std::path::Path; + +use atomic_core::change::format_v3::{ChangeReader, ChangeSignature, SectionType}; +use atomic_core::change::signing::{sign_change, verify_change_signature, ChangeSignatureError}; +use atomic_core::change::{Author, Change, ChangeHeader}; +use atomic_core::types::{Base32, Hash}; +use atomic_repository::record::{RecordOptions, SigningIdentity}; +use atomic_repository::Repository; +use tempfile::TempDir; + +fn add_file(repo: &Repository, repo_path: &Path, name: &str, content: &str) { + fs::write(repo_path.join(name), content).expect("write file"); + repo.add(name, Default::default()).expect("add file"); +} + +fn record_unsigned(repo: &Repository, message: &str) -> Hash { + let header = ChangeHeader::builder() + .message(message) + .author(Author::new("Test", Some("test@example.com"))) + .build(); + *repo + .record(header, RecordOptions::default()) + .expect("record") + .hash() +} + +fn signing_identity(name: &str) -> (SigningIdentity, [u8; 32]) { + // Deterministic test keypair — the seed IS the key material, so the + // test doesn't need the identity store. + let mut seed = [0u8; 32]; + for (i, b) in name.bytes().cycle().take(32).enumerate() { + seed[i] = b; + } + let signing_key = ed25519_dalek::SigningKey::from_bytes(&seed); + let public = signing_key.verifying_key().to_bytes(); + let pk_base32 = data_encoding::BASE32_NOPAD.encode(&public); + let did = format!("did:atomic:{}", &pk_base32[..16]); + ( + SigningIdentity { + signer_did: did, + secret_key: seed, + }, + public, + ) +} + +fn record_signed(repo: &Repository, message: &str, signing: &SigningIdentity) -> (Hash, Change) { + let header = ChangeHeader::builder() + .message(message) + .author(Author::new("Test", Some("test@example.com"))) + .build(); + let outcome = repo + .record( + header, + RecordOptions::default().with_signing_identity(signing.clone()), + ) + .expect("record"); + let hash = *outcome.hash(); + let change = repo.load_change(&hash).expect("load change"); + (hash, change) +} + +// ═══════════════════════════════════════════════════════════════════════ +// AC-1: signed records verify; signature does not change the hash +// ═══════════════════════════════════════════════════════════════════════ + +#[test] +fn signed_record_verifies_against_public_key() { + let temp = TempDir::new().unwrap(); + let repo_path = temp.path().to_path_buf(); + let repo = Repository::init(&repo_path).expect("init"); + + add_file(&repo, &repo_path, "signed.txt", "signed content\n"); + let (signing, public) = signing_identity("alice"); + let (hash, change) = record_signed(&repo, "signed change", &signing); + + // The change carries a signature. + let signature = change.signature.as_ref().expect("signature present"); + assert!(signature.is_ed25519()); + + // The signature verifies against the OUT-OF-BAND public key. + verify_change_signature(signature, &public, &hash).expect("verify"); + + // The stored .change file on disk parses and its SIGNATURE section + // round-trips. + let base32 = hash.to_base32(); + let file_path = repo_path + .join(".atomic/changes") + .join(&base32[..2]) + .join(format!("{}.change", base32)); + let bytes = fs::read(&file_path).expect("read .change file"); + + let mut cursor = std::io::Cursor::new(&bytes); + let mut reader = ChangeReader::open(&mut cursor).expect("open reader"); + let mut found_signature: Option = None; + while let Some(section) = reader.next_section().expect("section") { + if section.section_type == SectionType::Signature { + found_signature = Some( + section + .deserialize() + .expect("deserialize signature section"), + ); + } + } + let file_hash_bytes = reader.verify().expect("verify file hash"); + let file_hash = Hash::from_bytes(file_hash_bytes); + + let file_signature = found_signature.expect("SIGNATURE section present in file"); + assert_eq!(file_signature, *signature); + assert_eq!(file_hash, hash, "file hash equals change identity"); + verify_change_signature(&file_signature, &public, &file_hash).expect("verify from file"); +} + +#[test] +fn signature_does_not_change_change_hash() { + // A signed change's hash is a pure function of content+header+deps: + // stripping the signature and re-serializing must yield the SAME hash, + // and re-signing with a DIFFERENT key must also yield the same hash. + let temp = TempDir::new().unwrap(); + let repo_path = temp.path().to_path_buf(); + let repo = Repository::init(&repo_path).expect("init"); + + add_file(&repo, &repo_path, "h.txt", "hash stability\n"); + let (signing_alice, _alice_public) = signing_identity("alice-signing"); + let (signing_mallory, _mallory_public) = signing_identity("mallory-signing"); + + let (_hash, change) = record_signed(&repo, "signed by alice", &signing_alice); + + // Strip the signature and re-serialize — the hash must not change. + let signed_hash = Change::hash(&change).expect("hash signed"); + let mut unsigned = change.clone(); + unsigned.signature = None; + let unsigned_hash_of_same = unsigned.hash().expect("hash unsigned variant"); + assert_eq!( + signed_hash, unsigned_hash_of_same, + "removing the signature must not change the change hash" + ); + + // Re-sign with a different key — hash still must not change. + let did = signing_mallory.signer_did.clone(); + unsigned + .sign_with(&did, &signing_mallory.secret_key, 999) + .expect("re-sign with mallory's key"); + let mallory_hash = unsigned.hash().expect("hash mallory-signed"); + assert_eq!( + signed_hash, mallory_hash, + "re-signing with a different key must not change the change hash" + ); +} + +// ═══════════════════════════════════════════════════════════════════════ +// AC-3: the forge scenario — embedded key is not a trust root +// ═══════════════════════════════════════════════════════════════════════ + +#[test] +fn forge_with_attacker_key_fails_against_victim_key() { + // An attacker signs content with THEIR key and claims to be the victim + // (their DID is embedded in the signature). Verification against the + // victim's known public key must FAIL. + let attacker = ed25519_dalek::SigningKey::from_bytes( + b"attacker-attacker-attacker-32byte!!"[..32] + .try_into() + .unwrap(), + ); + let victim = ed25519_dalek::SigningKey::from_bytes( + b"victim--victim--victim--victim--32b!!"[..32] + .try_into() + .unwrap(), + ); + + let hash = Hash::of(b"malicious content claiming to be Aaron"); + let attacker_did = "did:atomic:ATTACKERCLAIMINGTOBEVICTIM00000000"; + + // Attacker forges the signature section. + let forged = sign_change( + attacker_did, + attacker.to_bytes().as_slice().try_into().unwrap(), + &hash, + 1739290034, + ); + + // Verifier resolves the VICTIM's public key out-of-band (identity store, + // lookup-key, whatever) and checks. + let victim_public = victim.verifying_key().to_bytes(); + let err = verify_change_signature(&forged, &victim_public, &hash).unwrap_err(); + assert!( + matches!( + err, + ChangeSignatureError::SignatureVerificationFailed { .. } + ), + "forged signature must fail against victim's key, got: {err:?}" + ); + + // Sanity: it WOULD verify against the attacker's own key — proving that + // self-referential verification (checking against the embedded key) is + // broken and must never be used. + let attacker_public = attacker.verifying_key().to_bytes(); + verify_change_signature(&forged, &attacker_public, &hash).expect( + "forged sig verifies against attacker key — demonstrating why embedded-key trust is broken", + ); +} + +#[test] +fn unsigned_change_has_no_signature_and_still_works() { + // AC-5 (backward compat): an unsigned change loads fine; its signature + // is None; its content applies. + let temp = TempDir::new().unwrap(); + let repo_path = temp.path().to_path_buf(); + let repo = Repository::init(&repo_path).expect("init"); + + add_file(&repo, &repo_path, "legacy.txt", "legacy content\n"); + let hash = record_unsigned(&repo, "legacy unsigned change"); + + let change = repo.load_change(&hash).expect("load unsigned change"); + assert!( + change.signature.is_none(), + "unsigned change has no signature" + ); + + // Round-trip through serialize/deserialize still works. + let mut bytes = Vec::new(); + let re_hash = change.serialize(&mut bytes).expect("serialize"); + assert_eq!(re_hash, hash); + let (re_change, _) = + Change::deserialize(&mut std::io::Cursor::new(&bytes)).expect("deserialize"); + assert!(re_change.signature.is_none()); +} + +// ═══════════════════════════════════════════════════════════════════════ +// AC-5: signed and unsigned changes coexist in one repository +// ═══════════════════════════════════════════════════════════════════════ + +#[test] +fn signed_and_unsigned_changes_coexist() { + let temp = TempDir::new().unwrap(); + let repo_path = temp.path().to_path_buf(); + let repo = Repository::init(&repo_path).expect("init"); + + // Unsigned first (simulates a pre-signing change). + add_file(&repo, &repo_path, "old.txt", "old\n"); + let old = record_unsigned(&repo, "old unsigned"); + + // Then signed (new binary). + let (signing, public) = signing_identity("carol"); + add_file(&repo, &repo_path, "new.txt", "new\n"); + let new = { + let header = ChangeHeader::builder() + .message("new signed") + .author(Author::new("Test", Some("test@example.com"))) + .build(); + let outcome = repo + .record( + header, + RecordOptions::default().with_signing_identity(signing.clone()), + ) + .expect("record"); + *outcome.hash() + }; + + // Both load; only the new one carries a signature. + let old_change = repo.load_change(&old).expect("load old"); + let new_change = repo.load_change(&new).expect("load new"); + assert!(old_change.signature.is_none()); + assert!(new_change.signature.is_some()); + + // The signature verifies out-of-band. + verify_change_signature(new_change.signature.as_ref().unwrap(), &public, &new) + .expect("verify signed change"); + + // History shows both. + let history = repo + .log(atomic_repository::history::HistoryOptions::default()) + .expect("log"); + let hashes: Vec = history.into_iter().map(|e| e.hash).collect(); + assert!(hashes.contains(&old)); + assert!(hashes.contains(&new)); +} From 1779710a301bd5fc9da6acb569e1806165a4464b Mon Sep 17 00:00:00 2001 From: Sankalp Gilda Date: Sat, 19 Sep 2026 17:16:26 -0400 Subject: [PATCH 18/22] refactor(canonical): delegate JCS to a from-spec canonicalizer with vectors Replace the hand-written RFC 8785 walker in atomic-canonical/src/jcs.rs with a call into serde_json_canonicalizer. Member ordering and string escaping were already correct; number formatting was not. Section 3.2.2.3 requires the ECMAScript Number::toString algorithm, and serde_json's formatter crosses between decimal and exponent notation at different magnitudes and writes negative zero as -0.0, so two conforming implementations hashed the same logical document to different digests. The delegate formats through ryu_js, which is the variant the section names, and serializes through an explicit heap stack rather than the call stack. canonicalize keeps its signature, so no call site changes anywhere in the workspace. Eleven measured cases are pinned as fixtures under atomic-canonical/tests/vectors/ with a harness in tests/jcs_vectors.rs. Three conformance cases this entry point cannot decide are recorded in atomic-canonical/tests/vectors/INGEST-BOUNDARY.md rather than tested here: a repeated object member is gone before canonicalize is reached, and RFC 8785 admits both an integer past 2^53 and a fractional number. All of them want a strict decoder on the raw bytes at the boundary where documents arrive. --- Cargo.lock | 18 +++ atomic-canonical/Cargo.toml | 1 + atomic-canonical/src/jcs.rs | 81 +++++-------- atomic-canonical/tests/jcs_vectors.rs | 113 ++++++++++++++++++ .../tests/vectors/INGEST-BOUNDARY.md | 30 +++++ .../depth-at-the-cap-is-canonicalized.json | 8 ++ .../tests/vectors/number-decimal-2pow68.json | 8 ++ .../number-decimal-999999999999999700000.json | 8 ++ .../number-decimal-999999999999999900000.json | 8 ++ ...mber-decimal-below-exponent-threshold.json | 8 ++ ...number-exponent-1.0000000000000001e23.json | 8 ++ .../number-exponent-9.999999999999997e-7.json | 8 ++ .../number-exponent-9.999999999999997e22.json | 8 ++ .../number-negative-small-decimal.json | 8 ++ .../tests/vectors/number-negative-zero.json | 8 ++ .../vectors/number-rounded-to-its-double.json | 8 ++ 16 files changed, 280 insertions(+), 51 deletions(-) create mode 100644 atomic-canonical/tests/jcs_vectors.rs create mode 100644 atomic-canonical/tests/vectors/INGEST-BOUNDARY.md create mode 100644 atomic-canonical/tests/vectors/depth-at-the-cap-is-canonicalized.json create mode 100644 atomic-canonical/tests/vectors/number-decimal-2pow68.json create mode 100644 atomic-canonical/tests/vectors/number-decimal-999999999999999700000.json create mode 100644 atomic-canonical/tests/vectors/number-decimal-999999999999999900000.json create mode 100644 atomic-canonical/tests/vectors/number-decimal-below-exponent-threshold.json create mode 100644 atomic-canonical/tests/vectors/number-exponent-1.0000000000000001e23.json create mode 100644 atomic-canonical/tests/vectors/number-exponent-9.999999999999997e-7.json create mode 100644 atomic-canonical/tests/vectors/number-exponent-9.999999999999997e22.json create mode 100644 atomic-canonical/tests/vectors/number-negative-small-decimal.json create mode 100644 atomic-canonical/tests/vectors/number-negative-zero.json create mode 100644 atomic-canonical/tests/vectors/number-rounded-to-its-double.json diff --git a/Cargo.lock b/Cargo.lock index d61e1347..be8ff24c 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -169,6 +169,7 @@ dependencies = [ "data-encoding", "serde", "serde_json", + "serde_json_canonicalizer", "thiserror 1.0.69", ] @@ -2512,6 +2513,12 @@ version = "1.0.23" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "9774ba4a74de5f7b1c1451ed6cd5285a32eddb5cccb8cc655a4e50009e06477f" +[[package]] +name = "ryu-js" +version = "1.0.3" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "04d056b875a9d2e6cb9a61d127afee9ac5999b9f87bcb32079d1318e505be714" + [[package]] name = "same-file" version = "1.0.6" @@ -2633,6 +2640,17 @@ dependencies = [ "zmij", ] +[[package]] +name = "serde_json_canonicalizer" +version = "0.3.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "fe52319a927259afbfa5180c5157cd8167edfd3e8c254f9558c7fef44c5649f2" +dependencies = [ + "ryu-js", + "serde", + "serde_json", +] + [[package]] name = "serde_spanned" version = "0.6.9" diff --git a/atomic-canonical/Cargo.toml b/atomic-canonical/Cargo.toml index 58ceac78..c6c48ce5 100644 --- a/atomic-canonical/Cargo.toml +++ b/atomic-canonical/Cargo.toml @@ -16,4 +16,5 @@ bs58 = { workspace = true } chrono = { workspace = true } thiserror = { workspace = true } data-encoding = { workspace = true } +serde_json_canonicalizer = "0.3" atomic-identity = { workspace = true } diff --git a/atomic-canonical/src/jcs.rs b/atomic-canonical/src/jcs.rs index b1ec4d09..d0682a2b 100644 --- a/atomic-canonical/src/jcs.rs +++ b/atomic-canonical/src/jcs.rs @@ -9,63 +9,42 @@ //! hash (`hash.rs`) and the Data Integrity proof (`proof.rs`) go through //! `canonicalize`, so the two can never drift. //! -//! Object keys are sorted by UTF-16 code units as RFC 8785 §3.2.3 specifies -//! (not by UTF-8 bytes — the two differ once keys leave the BMP, e.g. an -//! emoji key sorts after `\u{ff61}` in UTF-8 but before it in UTF-16). +//! # The algorithm is delegated rather than written here //! -//! Scope note: numbers are emitted via `serde_json`'s formatter, which -//! matches RFC 8785 for the integer values our vocabulary admits; the full -//! ECMAScript number-to-string algorithm is only needed if floating-point -//! payloads are ever admitted. +//! The bytes come from `serde_json_canonicalizer`. Two of the three parts of +//! RFC 8785 are straightforward to write by hand and the third is not: +//! +//! * Object keys sort by UTF-16 code units as §3.2.3 specifies, not by UTF-8 +//! bytes — the two differ once a key leaves the BMP, e.g. an emoji key sorts +//! after `\u{ff61}` in UTF-8 but before it in UTF-16. +//! * Strings take the short form for the seven named escapes and lowercase +//! `\u00xx` for the rest of C0 (§3.2.2.2). +//! * **Numbers take the ECMAScript `Number::toString` algorithm** (§3.2.2.3), +//! which `serde_json`'s formatter is not. `serde_json` writes `-0.0` where the +//! algorithm writes `0`, keeps a `.0` on an integer-valued double where the +//! algorithm drops it, and crosses between decimal and exponent notation at +//! different magnitudes: `1e-6` for `0.000001`, `2.9514790517935283e+20` for +//! `295147905179352830000`. The delegate formats through `ryu_js`, which is +//! the ECMAScript variant the section names, and it serializes through an +//! explicit heap stack rather than the call stack. The cases that used to +//! diverge are pinned in `tests/vectors/`. +//! +//! Scope note: the entry point takes an already-parsed [`Value`], so faults that +//! only exist in the wire bytes cannot be decided here — a repeated object +//! member is gone before this function is called, and RFC 8785 admits an integer +//! past 2^53 by rounding it to its double. Those belong to a strict decoder at +//! the boundary where the bytes arrive; `tests/vectors/INGEST-BOUNDARY.md` names +//! the cases and what closes them. use serde_json::Value; /// Canonicalize a JSON value into its RFC-8785 string form. pub fn canonicalize(value: &Value) -> String { - let mut out = String::new(); - write_value(&mut out, value); - out -} - -fn write_value(out: &mut String, value: &Value) { - match value { - Value::Null => out.push_str("null"), - Value::Bool(true) => out.push_str("true"), - Value::Bool(false) => out.push_str("false"), - Value::Number(n) => out.push_str(&n.to_string()), - Value::String(s) => write_json_string(out, s), - Value::Array(items) => { - out.push('['); - for (i, item) in items.iter().enumerate() { - if i > 0 { - out.push(','); - } - write_value(out, item); - } - out.push(']'); - } - Value::Object(map) => { - let mut keys: Vec<&String> = map.keys().collect(); - keys.sort_by(|a, b| a.encode_utf16().cmp(b.encode_utf16())); - out.push('{'); - for (i, key) in keys.iter().enumerate() { - if i > 0 { - out.push(','); - } - write_json_string(out, key); - out.push(':'); - write_value(out, &map[*key]); - } - out.push('}'); - } - } -} - -/// Emit a JSON string with standard escaping. `serde_json` produces a valid, -/// minimally-escaped JSON string literal (quotes included), which matches JCS -/// for the ASCII content in our records. -fn write_json_string(out: &mut String, s: &str) { - out.push_str(&serde_json::to_string(s).expect("string serialization is infallible")); + // Infallible for a `Value`: there is no writer to fail against, every + // member name is already a Rust `String`, and the delegate's only other + // error path is a non-finite float, which `Value` cannot hold. This mirrors + // the expectation the hand-written string helper carried before it. + serde_json_canonicalizer::to_string(value).expect("canonicalizing a Value is infallible") } #[cfg(test)] diff --git a/atomic-canonical/tests/jcs_vectors.rs b/atomic-canonical/tests/jcs_vectors.rs new file mode 100644 index 00000000..02ccc69e --- /dev/null +++ b/atomic-canonical/tests/jcs_vectors.rs @@ -0,0 +1,113 @@ +//! Canonicalization vectors, one fixture per divergence. +//! +//! Each file in `tests/vectors/` carries an input document, the divergence it +//! pins, and the canonical form RFC 8785 requires. The number cases are Appendix +//! B rows: the canonical text of a double is one string and no other, so a +//! canonicalizer that writes `1e-6` where the algorithm writes `0.000001` +//! produces a different hash for the same value, and a second implementation +//! then rejects a proof this one accepts. +//! +//! `tests/vectors/INGEST-BOUNDARY.md` lists the conformance cases this entry +//! point cannot decide, because it receives an already-parsed value rather than +//! the bytes. + +use std::fs; +use std::path::{Path, PathBuf}; + +use atomic_canonical::jcs; +use serde_json::Value; + +fn vector_dir() -> PathBuf { + Path::new(env!("CARGO_MANIFEST_DIR")).join("tests/vectors") +} + +fn fixture(name: &str) -> Value { + let path = vector_dir().join(format!("{name}.json")); + let raw = fs::read_to_string(&path).unwrap_or_else(|e| panic!("read {path:?}: {e}")); + serde_json::from_str(&raw).unwrap_or_else(|e| panic!("parse {path:?}: {e}")) +} + +/// Parse the fixture's input the way any caller receiving JSON does, canonicalize +/// it, and require the exact bytes. +fn assert_canonical(name: &str) { + let f = fixture(name); + let text = f["input"].as_str().expect("fixture carries an input string"); + let value: Value = serde_json::from_str(text).expect("fixture input is valid JSON"); + let expected = f["expect"]["canonical"] + .as_str() + .expect("fixture expects a canonical form"); + assert_eq!( + jcs::canonicalize(&value), + expected, + "{name} ({}): {}", + f["vector"].as_str().unwrap_or("?"), + f["divergence"].as_str().unwrap_or("") + ); +} + +#[test] +fn number_decimal_below_exponent_threshold() { + assert_canonical("number-decimal-below-exponent-threshold"); +} + +#[test] +fn number_decimal_2pow68() { + assert_canonical("number-decimal-2pow68"); +} + +#[test] +fn number_decimal_999999999999999700000() { + assert_canonical("number-decimal-999999999999999700000"); +} + +#[test] +fn number_decimal_999999999999999900000() { + assert_canonical("number-decimal-999999999999999900000"); +} + +#[test] +fn number_negative_small_decimal() { + assert_canonical("number-negative-small-decimal"); +} + +#[test] +fn number_negative_zero() { + assert_canonical("number-negative-zero"); +} + +#[test] +fn number_exponent_9_999999999999997e22() { + assert_canonical("number-exponent-9.999999999999997e22"); +} + +#[test] +fn number_exponent_1_0000000000000001e23() { + assert_canonical("number-exponent-1.0000000000000001e23"); +} + +#[test] +fn number_exponent_9_999999999999997e_minus_7() { + assert_canonical("number-exponent-9.999999999999997e-7"); +} + +#[test] +fn number_rounded_to_its_double() { + assert_canonical("number-rounded-to-its-double"); +} + +/// The accept half of the depth condition: a document at 128 containers +/// canonicalizes, and the delegate walks it on the heap rather than the call +/// stack. The refusal half needs a fallible boundary -- see +/// `tests/vectors/INGEST-BOUNDARY.md`. +#[test] +fn depth_at_the_cap_is_canonicalized() { + let f = fixture("depth-at-the-cap-is-canonicalized"); + assert!(f["input_generated"].as_str().is_some()); + + let mut value = Value::Null; + for _ in 0..128 { + value = Value::Array(vec![value]); + } + let canonical = jcs::canonicalize(&value); + assert_eq!(canonical, format!("{}null{}", "[".repeat(128), "]".repeat(128))); +} diff --git a/atomic-canonical/tests/vectors/INGEST-BOUNDARY.md b/atomic-canonical/tests/vectors/INGEST-BOUNDARY.md new file mode 100644 index 00000000..20f86019 --- /dev/null +++ b/atomic-canonical/tests/vectors/INGEST-BOUNDARY.md @@ -0,0 +1,30 @@ +# Vectors this entry point cannot decide + +`jcs::canonicalize` takes an already-parsed `serde_json::Value`. Three conformance +vectors ask for a refusal that no function of that shape can give, because the +fault either vanished during the parse or is not a fault under RFC 8785 at all. +They are listed here rather than dropped, so the follow-up has its brief in-tree. + +| vector | condition | what it asks for | why not here | +| --- | --- | --- | --- | +| `v0f4f2093061d303f` | duplicate member | reject `{"a":1,"a":2}` | `serde_json` keeps one of the two members while parsing, so the repeat is gone before `canonicalize` is called. Measured: the document canonicalizes to `{"a":2}`, and `{"a":2,"a":1}` to `{"a":1}` -- two wire documents, two canonical forms, and a signature over either verifies. | +| `v679f56481420e45a` | unsafe integer | reject `9007199254740993` | RFC 8785 defers number formatting to ECMAScript, which has one numeric type, so the specification *admits* the token and writes the double it rounds to. Refusing it is the RFC 7493 I-JSON profile, which is a stricter profile rather than RFC 8785 itself. Pinned as an accept in `number-rounded-to-its-double.json`. | +| `v97f5d8777e514257` | non-integer in a signed field | reject `0.7` | Same shape: RFC 8785 admits a fractional number. Refusing one is a field-level profile decision, not a canonicalization rule. | + +A fourth case is half-covered. `vd94ac70c9f0d84bf` asks a canonicalizer to refuse +a document nested one container past a stated cap with a catchable error. The +accept half is pinned in `depth-at-the-cap-is-canonicalized.json`; the refusal +half needs a fallible boundary, and an infallible `canonicalize(&Value) -> String` +has nowhere to put it. + +## What closes all four + +A strict decoder on the raw bytes, ahead of this function: it refuses a repeated +member, caps nesting at 128 with an error rather than a stack walk, refuses a +string that is not a sequence of Unicode scalar values, and optionally applies the +I-JSON safe-integer profile. `jcs-admit` on crates.io does exactly that and then +hands canonical output to the same `serde_json_canonicalizer` this file already +uses, so adopting it adds refusals without changing a single byte of output. + +The place it belongs is wherever a document arrives as bytes rather than as a +value built in-process. diff --git a/atomic-canonical/tests/vectors/depth-at-the-cap-is-canonicalized.json b/atomic-canonical/tests/vectors/depth-at-the-cap-is-canonicalized.json new file mode 100644 index 00000000..c0a3a890 --- /dev/null +++ b/atomic-canonical/tests/vectors/depth-at-the-cap-is-canonicalized.json @@ -0,0 +1,8 @@ +{ + "vector": "vd94ac70c9f0d84bf (accept half)", + "divergence": "a document nested 128 containers deep canonicalizes, which is the accept side of the depth condition", + "input_generated": "128 nested arrays around null", + "expect": { + "canonical_shape": "128 open brackets, null, 128 close brackets" + } +} diff --git a/atomic-canonical/tests/vectors/number-decimal-2pow68.json b/atomic-canonical/tests/vectors/number-decimal-2pow68.json new file mode 100644 index 00000000..57d98b14 --- /dev/null +++ b/atomic-canonical/tests/vectors/number-decimal-2pow68.json @@ -0,0 +1,8 @@ +{ + "vector": "v362600d69975d14c", + "divergence": "decimal-to-exponent threshold: a magnitude below 1e21 stays decimal", + "input": "{\"n\":295147905179352830000}", + "expect": { + "canonical": "{\"n\":295147905179352830000}" + } +} diff --git a/atomic-canonical/tests/vectors/number-decimal-999999999999999700000.json b/atomic-canonical/tests/vectors/number-decimal-999999999999999700000.json new file mode 100644 index 00000000..11a1c1b0 --- /dev/null +++ b/atomic-canonical/tests/vectors/number-decimal-999999999999999700000.json @@ -0,0 +1,8 @@ +{ + "vector": "vaa33c34f7bd1058b", + "divergence": "decimal notation retained just below the 1e21 threshold", + "input": "{\"n\":999999999999999700000}", + "expect": { + "canonical": "{\"n\":999999999999999700000}" + } +} diff --git a/atomic-canonical/tests/vectors/number-decimal-999999999999999900000.json b/atomic-canonical/tests/vectors/number-decimal-999999999999999900000.json new file mode 100644 index 00000000..6c188bf1 --- /dev/null +++ b/atomic-canonical/tests/vectors/number-decimal-999999999999999900000.json @@ -0,0 +1,8 @@ +{ + "vector": "vb177ef3b3a945a72", + "divergence": "decimal notation retained for the last double below 1e21", + "input": "{\"n\":999999999999999900000}", + "expect": { + "canonical": "{\"n\":999999999999999900000}" + } +} diff --git a/atomic-canonical/tests/vectors/number-decimal-below-exponent-threshold.json b/atomic-canonical/tests/vectors/number-decimal-below-exponent-threshold.json new file mode 100644 index 00000000..96129517 --- /dev/null +++ b/atomic-canonical/tests/vectors/number-decimal-below-exponent-threshold.json @@ -0,0 +1,8 @@ +{ + "vector": "vaf1b67f038ffff84", + "divergence": "decimal-to-exponent threshold: 1e-6 is written in decimal notation", + "input": "{\"n\":0.000001}", + "expect": { + "canonical": "{\"n\":0.000001}" + } +} diff --git a/atomic-canonical/tests/vectors/number-exponent-1.0000000000000001e23.json b/atomic-canonical/tests/vectors/number-exponent-1.0000000000000001e23.json new file mode 100644 index 00000000..21bb9d5e --- /dev/null +++ b/atomic-canonical/tests/vectors/number-exponent-1.0000000000000001e23.json @@ -0,0 +1,8 @@ +{ + "vector": "vb7dd5fb8dcb6e345", + "divergence": "seventeen significant digits kept where sixteen would not round-trip", + "input": "{\"n\":1.0000000000000001e+23}", + "expect": { + "canonical": "{\"n\":1.0000000000000001e+23}" + } +} diff --git a/atomic-canonical/tests/vectors/number-exponent-9.999999999999997e-7.json b/atomic-canonical/tests/vectors/number-exponent-9.999999999999997e-7.json new file mode 100644 index 00000000..4f2e4350 --- /dev/null +++ b/atomic-canonical/tests/vectors/number-exponent-9.999999999999997e-7.json @@ -0,0 +1,8 @@ +{ + "vector": "vf2f6991c55096b1a", + "divergence": "shortest round-tripping digits below the 1e-6 threshold", + "input": "{\"n\":9.999999999999997e-7}", + "expect": { + "canonical": "{\"n\":9.999999999999997e-7}" + } +} diff --git a/atomic-canonical/tests/vectors/number-exponent-9.999999999999997e22.json b/atomic-canonical/tests/vectors/number-exponent-9.999999999999997e22.json new file mode 100644 index 00000000..50f2ebfa --- /dev/null +++ b/atomic-canonical/tests/vectors/number-exponent-9.999999999999997e22.json @@ -0,0 +1,8 @@ +{ + "vector": "v843566cf72cd33c3", + "divergence": "shortest round-tripping digits in exponent notation", + "input": "{\"n\":9.999999999999997e+22}", + "expect": { + "canonical": "{\"n\":9.999999999999997e+22}" + } +} diff --git a/atomic-canonical/tests/vectors/number-negative-small-decimal.json b/atomic-canonical/tests/vectors/number-negative-small-decimal.json new file mode 100644 index 00000000..afc61d9d --- /dev/null +++ b/atomic-canonical/tests/vectors/number-negative-small-decimal.json @@ -0,0 +1,8 @@ +{ + "vector": "vd2659c36ece7eb47", + "divergence": "a negative magnitude above 1e-6 stays decimal", + "input": "{\"n\":-0.0000033333333333333333}", + "expect": { + "canonical": "{\"n\":-0.0000033333333333333333}" + } +} diff --git a/atomic-canonical/tests/vectors/number-negative-zero.json b/atomic-canonical/tests/vectors/number-negative-zero.json new file mode 100644 index 00000000..ad01f931 --- /dev/null +++ b/atomic-canonical/tests/vectors/number-negative-zero.json @@ -0,0 +1,8 @@ +{ + "vector": "v9a8b364b8bc121de", + "divergence": "negative zero is written as 0, and -0.0 is not valid ECMAScript output at all", + "input": "{\"n\":-0.0}", + "expect": { + "canonical": "{\"n\":0}" + } +} diff --git a/atomic-canonical/tests/vectors/number-rounded-to-its-double.json b/atomic-canonical/tests/vectors/number-rounded-to-its-double.json new file mode 100644 index 00000000..c23749de --- /dev/null +++ b/atomic-canonical/tests/vectors/number-rounded-to-its-double.json @@ -0,0 +1,8 @@ +{ + "vector": "v679f56481420e45a", + "divergence": "RFC 8785 treats every number as a double, so 2^53+1 canonicalizes to the double it rounds to", + "input": "{\"n\":9007199254740993}", + "expect": { + "canonical": "{\"n\":9007199254740992}" + } +} From 695c6c72c792c50a4e38cc16565ad74fc94b4699 Mon Sep 17 00:00:00 2001 From: Sankalp Gilda Date: Fri, 25 Sep 2026 12:27:32 -0400 Subject: [PATCH 19/22] style(canonical): apply rustfmt to the JCS vector tests The Format job runs cargo fmt --all -- --check, which rejected two statements in atomic-canonical/tests/jcs_vectors.rs. No behaviour change. --- atomic-canonical/tests/jcs_vectors.rs | 9 +++++++-- 1 file changed, 7 insertions(+), 2 deletions(-) diff --git a/atomic-canonical/tests/jcs_vectors.rs b/atomic-canonical/tests/jcs_vectors.rs index 02ccc69e..3b668362 100644 --- a/atomic-canonical/tests/jcs_vectors.rs +++ b/atomic-canonical/tests/jcs_vectors.rs @@ -31,7 +31,9 @@ fn fixture(name: &str) -> Value { /// it, and require the exact bytes. fn assert_canonical(name: &str) { let f = fixture(name); - let text = f["input"].as_str().expect("fixture carries an input string"); + let text = f["input"] + .as_str() + .expect("fixture carries an input string"); let value: Value = serde_json::from_str(text).expect("fixture input is valid JSON"); let expected = f["expect"]["canonical"] .as_str() @@ -109,5 +111,8 @@ fn depth_at_the_cap_is_canonicalized() { value = Value::Array(vec![value]); } let canonical = jcs::canonicalize(&value); - assert_eq!(canonical, format!("{}null{}", "[".repeat(128), "]".repeat(128))); + assert_eq!( + canonical, + format!("{}null{}", "[".repeat(128), "]".repeat(128)) + ); } From 63e5f79de419230bd1cd6a86f67be32478e25499 Mon Sep 17 00:00:00 2001 From: Sankalp Gilda Date: Wed, 30 Sep 2026 04:50:24 -0400 Subject: [PATCH 20/22] fix(canonical): refuse to store a grant its own readers would refuse Every reader of a stored grant admits it under the I-JSON profile before verifying anything: load_for_delegate, and grant list, verify, push and revoke. The writers did not. mint signs a maxChanges of 2^53 or more, or a Unicode noncharacter in a name or description, and the grant is then stored, exported or printed and skipped by every reader on the machine. Add delegation::encode_for_storage, which renders the indented form the store has always held and admits those bytes before returning them, and use it in grant new, agent create and agent renew. agent create now encodes before it saves the agent identity, so a refusal leaves no identity behind without its certificate. Tests: the largest safe count is stored and found by load_for_delegate; 2^53 and a noncharacter description are refused with the reader's own fault. Deleting the admission call in encode_for_storage fails exactly the two refusal tests. --- atomic-canonical/src/delegation.rs | 27 ++++++++ atomic-canonical/tests/ingest_boundary.rs | 63 +++++++++++++++++++ .../src/commands/identity/agent/create.rs | 9 ++- .../src/commands/identity/agent/renew.rs | 7 ++- atomic-cli/src/commands/identity/delegate.rs | 9 ++- 5 files changed, 106 insertions(+), 9 deletions(-) diff --git a/atomic-canonical/src/delegation.rs b/atomic-canonical/src/delegation.rs index 31bc9e9d..c3929bb2 100644 --- a/atomic-canonical/src/delegation.rs +++ b/atomic-canonical/src/delegation.rs @@ -294,6 +294,33 @@ pub fn encode_for_transport(document: &Value) -> String { data_encoding::BASE64URL_NOPAD.encode(canonical.as_bytes()) } +/// Render a minted certificate for the identity store, admitting the bytes it +/// produces before handing them back. +/// +/// Every reader of a stored grant admits it under the profile in +/// [`crate::jcs`] before it verifies anything: [`load_for_delegate`], and the +/// CLI's `grant list`, `grant verify`, `grant push` and `grant revoke`. So a +/// certificate written outside that profile is signed, stored and then skipped +/// by every one of them. [`mint`] takes a `u64` count and free-text names, and +/// both can leave the profile: a `maxChanges` of 2^53 or more, or a Unicode +/// noncharacter in a name or description. Refusing here, before anything is +/// stored, exported or printed, keeps what this crate writes and what it reads +/// back the same set of documents. +/// +/// The bytes returned are the indented form the store has always held, so a +/// certificate that passes is stored exactly as before. +/// +/// # Errors +/// +/// [`CanonicalError::Admission`] naming the fault when the reader would refuse +/// the document, or [`CanonicalError::Proof`] if it does not serialize. +pub fn encode_for_storage(document: &Value) -> Result { + let rendered = serde_json::to_string_pretty(document) + .map_err(|e| CanonicalError::Proof(format!("certificate does not serialize: {e}")))?; + jcs::admit_document(rendered.as_bytes())?; + Ok(rendered) +} + /// Decode a certificate presented in a request header. /// /// Checks the size cap first, then base64, then admission. Does **not** verify — diff --git a/atomic-canonical/tests/ingest_boundary.rs b/atomic-canonical/tests/ingest_boundary.rs index a5c3cf1f..f4aeb74f 100644 --- a/atomic-canonical/tests/ingest_boundary.rs +++ b/atomic-canonical/tests/ingest_boundary.rs @@ -249,3 +249,66 @@ fn load_for_delegate_skips_a_stored_document_with_a_repeated_member() { "the admissible certificate survives and the repeated member does not" ); } + +// --------------------------------------------------------------------------- +// The writer holds the readers' profile +// --------------------------------------------------------------------------- + +fn certificate_with(p: &Pair, max_changes: u64, description: Option<&str>) -> Value { + let mut scope = DelegationScope::builder() + .permission(atomic_identity::delegation::DelegationPermission::Record) + .project("acme/*") + .max_changes(max_changes); + if let Some(text) = description { + scope = scope.description(text); + } + let terms = Delegation::new(&p.human, &p.agent, scope.build()); + delegation::mint(&p.human, &p.human_key, &terms) +} + +/// The largest count the admission profile takes is stored, and it reads back +/// through the same boundary every stored grant passes on the way in. +#[test] +fn the_largest_safe_count_is_stored_and_read_back() { + let root = tempfile::tempdir().expect("temp store"); + let store = IdentityStore::open(root.path()).expect("open store"); + let p = pair(); + let doc = certificate_with(&p, 9_007_199_254_740_991, None); + let stored = delegation::encode_for_storage(&doc).expect("2^53 - 1 is a safe integer"); + let parsed = delegation::parse(&doc).expect("parse the minted certificate"); + store + .save_delegation(&parsed.id.to_base32(), &stored) + .expect("save"); + + let found = delegation::load_for_delegate(&store, &p.agent).expect("load"); + assert_eq!( + found.len(), + 1, + "a grant the writer stored is a grant the reader finds" + ); +} + +/// One past it. `mint` signs a `maxChanges` of 2^53 without complaint, and every +/// reader of the store then refuses the document, so `load_for_delegate` finds +/// nothing. The writer refuses it first, naming the same fault. +#[test] +fn a_count_the_readers_refuse_is_not_stored() { + let p = pair(); + let doc = certificate_with(&p, 9_007_199_254_740_992, None); + match delegation::encode_for_storage(&doc) { + Err(CanonicalError::Admission(Admission::UnsafeInteger { .. })) => {} + other => panic!("expected an unsafe-integer refusal, got {other:?}"), + } +} + +/// The same for text: a Unicode noncharacter in a description is signed by +/// `mint` and refused by every reader, so the writer refuses it too. +#[test] +fn a_noncharacter_the_readers_refuse_is_not_stored() { + let p = pair(); + let doc = certificate_with(&p, 64, Some("scratch \u{FDD0} lane")); + match delegation::encode_for_storage(&doc) { + Err(CanonicalError::Admission(Admission::StringNotScalar { .. })) => {} + other => panic!("expected a noncharacter refusal, got {other:?}"), + } +} diff --git a/atomic-cli/src/commands/identity/agent/create.rs b/atomic-cli/src/commands/identity/agent/create.rs index ca6fe0c0..283229dd 100644 --- a/atomic-cli/src/commands/identity/agent/create.rs +++ b/atomic-cli/src/commands/identity/agent/create.rs @@ -188,6 +188,12 @@ impl Create { )) })?; let certificate = cert::mint(&delegator, &delegator_keypair, &terms); + // Admitted before anything is persisted, so a refusal leaves no agent + // identity behind without the certificate that was meant to cover it. + let document = + cert::encode_for_storage(&certificate).map_err(|e| CliError::InvalidArgument { + message: format!("this grant would be refused when it is read back: {e}"), + })?; // 5. Persist. The identity and its key first — a certificate naming a // key that was never stored is worse than no certificate. @@ -195,9 +201,6 @@ impl Create { .save_with_keypair(&agent, &keypair, None) .map_err(|e| CliError::Internal(anyhow::anyhow!("Failed to save identity: {e}")))?; - let document = serde_json::to_string_pretty(&certificate).map_err(|e| { - CliError::Internal(anyhow::anyhow!("Failed to encode certificate: {e}")) - })?; let delegation_id = terms.id.to_base32(); store .save_delegation(&delegation_id, &document) diff --git a/atomic-cli/src/commands/identity/agent/renew.rs b/atomic-cli/src/commands/identity/agent/renew.rs index 3822645c..7552fb7f 100644 --- a/atomic-cli/src/commands/identity/agent/renew.rs +++ b/atomic-cli/src/commands/identity/agent/renew.rs @@ -145,9 +145,10 @@ impl Renew { })?; let certificate = cert::mint(&delegator, &keypair, &terms); - let document = serde_json::to_string_pretty(&certificate).map_err(|e| { - CliError::Internal(anyhow::anyhow!("Failed to encode certificate: {e}")) - })?; + let document = + cert::encode_for_storage(&certificate).map_err(|e| CliError::InvalidArgument { + message: format!("this grant would be refused when it is read back: {e}"), + })?; store .save_delegation(&terms.id.to_base32(), &document) .map_err(|e| CliError::Internal(anyhow::anyhow!("Failed to store certificate: {e}")))?; diff --git a/atomic-cli/src/commands/identity/delegate.rs b/atomic-cli/src/commands/identity/delegate.rs index a9e3f4fb..68d43946 100644 --- a/atomic-cli/src/commands/identity/delegate.rs +++ b/atomic-cli/src/commands/identity/delegate.rs @@ -178,9 +178,12 @@ impl Command for Delegate { )) })?; let certificate = cert::mint(&delegator, &keypair, &terms); - let document = serde_json::to_string_pretty(&certificate).map_err(|e| { - CliError::Internal(anyhow::anyhow!("Failed to encode certificate: {e}")) - })?; + // Admitted before it is stored, exported or printed: a grant every + // reader on this machine would refuse is not worth signing. + let document = + cert::encode_for_storage(&certificate).map_err(|e| CliError::InvalidArgument { + message: format!("this grant would be refused when it is read back: {e}"), + })?; // `--export` writes the wire form and nothing else, so the output is // safe to capture in a shell substitution. From 3cdeb28d29beae2bd48ad7a057389fd61688efe2 Mon Sep 17 00:00:00 2001 From: Sankalp Gilda Date: Wed, 30 Sep 2026 04:52:19 -0400 Subject: [PATCH 21/22] docs(canonical): stop saying load_for_delegate logs each skip The comment promised a warn log for every stored certificate the function skips, and there is no logging call in it or a logging dependency in the crate. Say what happens instead. --- atomic-canonical/src/delegation.rs | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/atomic-canonical/src/delegation.rs b/atomic-canonical/src/delegation.rs index c3929bb2..3f650ac1 100644 --- a/atomic-canonical/src/delegation.rs +++ b/atomic-canonical/src/delegation.rs @@ -541,8 +541,8 @@ impl StoredDelegation { /// Certificates that fail admission or verification are **skipped**, not /// returned as errors. This is the one place a corrupt or foreign file in the /// store could otherwise take down every agent operation, and a certificate that -/// does not verify has no authority to convey in any case. Each skip is logged -/// at warn. +/// does not verify has no authority to convey in any case. A skip is not +/// logged: this crate has no logging dependency. /// /// Verification is self-contained — it uses the delegator key the certificate /// carries — so this works on a machine that holds only the agent's key. From 9fa8a31259ff03bbd07acafc6231bc10dbb916e9 Mon Sep 17 00:00:00 2001 From: Sankalp Gilda Date: Wed, 30 Sep 2026 05:11:25 -0400 Subject: [PATCH 22/22] test(canonical): fail the build if a Value can carry an unformattable number canonicalize expects serde_json_canonicalizer never to fail on a Value. That holds only while serde_json's arbitrary_precision feature is off: with it on, a Value keeps 1e400 as text, the delegate parses it to infinity and errors, and the expect panics on input anyone can supply. Features unify across the workspace, so any new dependency could turn it on. Pin the premise with a test. --- atomic-canonical/src/jcs.rs | 15 +++++++++++++++ 1 file changed, 15 insertions(+) diff --git a/atomic-canonical/src/jcs.rs b/atomic-canonical/src/jcs.rs index 53893936..b5c50b80 100644 --- a/atomic-canonical/src/jcs.rs +++ b/atomic-canonical/src/jcs.rs @@ -170,4 +170,19 @@ mod tests { "supplementary-plane keys must sort by UTF-16 code units" ); } + + /// `canonicalize` expects the delegate never to fail on a `Value`. That + /// holds because a `Value` cannot carry a number the delegate refuses to + /// format, and it holds only while `serde_json`'s `arbitrary_precision` + /// feature is off: with it on, a `Value` keeps `1e400` as text, the delegate + /// parses that text to infinity and returns an error, and the `expect` + /// panics on input anyone can supply. Features unify across the workspace, + /// so any dependency could turn it on. This fails the build the day one does. + #[test] + fn a_value_cannot_carry_a_number_the_delegate_refuses() { + assert!( + serde_json::from_str::("1e400").is_err(), + "serde_json's arbitrary_precision feature is on, so canonicalize can panic" + ); + } }