//! The open/create/commit driver: the cold-open path and the atomic write //! protocol (Chapter 8 §"The Atomic Write Protocol", §"Crash Recovery", //! §"Streaming Reads"). //! //! A [`Bundle`] wraps a [`BlockStore`] and the currently-active prelude state. //! Its two load-bearing operations are: //! //! * [`Bundle::open`] — the cold-open procedure: header → superblock selection → //! manifest read-and-verify, exposing the canonical roots for streaming. //! * [`Bundle::commit`] — the seven-step atomic write protocol, whose final //! durable flush is the commit point. A crash at any step leaves the bundle at //! the previous generation (the active superblock and its reachable chunks are //! never touched) or, only after the final flush, at the new generation. use crate::block; use crate::chunk::{ chunk_content_hash, chunk_id, content_hash_for, ChunkKind, ChunkRef, CompressionAlgorithm, }; use crate::codec::DecodeError; use crate::error::{BundleError, IntegrityAnomaly}; use crate::header::{FixedHeader, FormatEpoch, SLOT_A_OFFSET, SLOT_B_OFFSET}; use crate::ids::{BlobId, FileUuid, ReductionAlgorithmVersion, SchemaVersion, WallClockTime}; use crate::manifest::{BlobRef, Manifest, ProfileDeclaration}; use crate::opindex::OperationIndex; use crate::store::{read_vec, BlockStore, MemStore}; use crate::superblock::{ select_active, CommitState, ProfileId, Slot, SlotParse, SlotReject, Superblock, SUPERBLOCK_LEN, }; use epiphany_determinism::{ChunkId, ContentHash}; use std::collections::BTreeMap; /// Offset where variable-length body content begins (after the header and both /// superblock slots): 576 bytes. pub const BODY_START: u64 = SLOT_B_OFFSET + SUPERBLOCK_LEN; /// The schema major version this reader can parse for a **generic canonical /// chunk** (Chapter 8 §"Schema Versioning"). A chunk at a higher major is not /// interpretable by this reader as canonical state. /// /// This is the baseline for a **generic** chunk role. Admission of major 1 is /// raised **per chunk role** by `max_supported_major` as each role's versioned /// path lands — never as a blanket accept-set ahead of the decoders (which would /// let a spec-valid major-1 chunk reach an unversioned decoder and be mis-read). /// The operation-envelope-block role is raised to major 1 (schema-major track /// D2: a block bearing a v1 `CreateRegion`); the manifest and layout-cache roles /// stay at `0`. pub const SUPPORTED_SCHEMA_MAJOR: u16 = 0; /// The maximum schema major this reader admits for a chunk of `kind` — the upper /// bound of its per-role accept-set `[0, max]` (Binary Format companion /// §"Schema Major 1", "The accept-set gate"). /// /// `OperationEnvelopeBlock` admits major 3 (schema major 2 fills the /// cross-cutting/staff/metadata bodies its payloads embed; major 1 embedded a /// v1 `CreateRegion`; the reader treats the block bytes opaquely, so it /// parses a higher-major block without decoding the payload). **Schema major /// 3 is raised by genesis tranche G2b** (`spec/CONTRACT_GENESIS_G2B_TUNING.md`): /// `SetTuningContext` is the sole operation payload that embeds the tuning /// context (`epiphany_core::TuningContextSettings`, born at major 3 /// unconditionally), so an op block carrying one is now born at v3 — this /// superseded the earlier Push 4b tranche 3b-i note claiming no payload ever /// would. `Snapshot` admits major 3 for the acceleration full-`Score` form /// (decoded through the core versioned seam); the canonical BASE carried /// under the same kind must stay major 0, enforced per role. Every other role /// stays at [`SUPPORTED_SCHEMA_MAJOR`] until its own versioned path lands — /// the layout cache, the operation index, and the manifest (carried opaquely, /// never grows a versioned layout). A chunk above its role's max is not /// admitted; for a **canonical** role that means the bundle opens read-only /// (a lower-major-only reader meeting a newer op block), not a hard reject. pub fn max_supported_major(kind: ChunkKind) -> u16 { match kind { ChunkKind::OperationEnvelopeBlock => 3, // The payload-polymorphic Snapshot role: the acceleration // full-`Score` form is decoded through the core versioned seam // (`Score::decode_canonical_versioned`, majors {0,1,2,3}). The // *canonical base* must stay major 0 regardless — that is enforced // per ROLE (`mis_stamped_canonical_base`, consulted at open and // commit), not by this per-kind bound. ChunkKind::Snapshot => 3, _ => SUPPORTED_SCHEMA_MAJOR, } } /// The conformance-profile major version this implementation understands /// (Chapter 8 §"Format Profiles"). A profile declared at a higher major is a /// future capability set this reader cannot honor. pub const SUPPORTED_PROFILE_MAJOR: u32 = 0; /// Reader resource limit on a manifest chunk. The spec calls manifests "a few /// kilobytes at most"; this generous bound stops an untrusted superblock length /// from driving an allocation before the bytes are read (finding: untrusted /// length → OOM/truncation). pub const MAX_MANIFEST_BYTES: u64 = 64 << 20; /// Reader resource limit on a single chunk read, bounding allocation from an /// untrusted length regardless of (possibly sparse) file size. pub const MAX_CHUNK_BYTES: u64 = 256 << 20; /// Default reader resource limit on a blob (Chapter 8 §"Blobs"). A `BlobRef`'s /// own `declared_max_uncompressed_length`, if smaller, also applies. pub const MAX_BLOB_BYTES: u64 = 1 << 30; /// A chunk to be written by a commit: an opaque payload plus its kind and schema /// version. The bundle assigns its offset, computes its content hash, and /// returns a [`ChunkRef`] for the manifest builder to wire into roots. #[derive(Clone, Debug)] pub struct StagedChunk { /// The chunk's kind (dispatches parsing on read). pub kind: ChunkKind, /// The schema version the payload is encoded against. pub schema_version: SchemaVersion, /// The opaque uncompressed payload bytes. pub payload: Vec, } impl StagedChunk { /// A staged operation-envelope block at schema major 0 (the baseline: no /// operation in the block carries a versioned payload). Use /// [`StagedChunk::operation_block_versioned`] for a block whose operations /// may carry a higher-major payload (a v1 `CreateRegion`; a v2 /// cross-cutting/staff/metadata value under minimal stamping). pub fn operation_block(payload: Vec) -> Self { StagedChunk::operation_block_versioned(payload, SchemaVersion::V0) } /// A staged operation-envelope block at the given schema version. The version /// is the maximum over the block's operations /// (`OperationEnvelope::schema_major`, projected to `SchemaVersion`): a block /// bearing a v1 `CreateRegion` is stamped [`SchemaVersion::V1`], so a /// major-0-only reader opens the bundle read-only rather than /// mis-parsing v1 op bytes as v0 (Binary Format companion §"Schema Major 1"). pub fn operation_block_versioned(payload: Vec, schema_version: SchemaVersion) -> Self { StagedChunk { kind: ChunkKind::OperationEnvelopeBlock, schema_version, payload, } } /// A staged operation-index chunk at the current schema version (Chapter 8 /// §"The Operation Index") — a non-canonical accelerator; the payload is an /// [`OperationIndex::encode`](crate::OperationIndex::encode). v0 writes it /// uncompressed, like every chunk (the spec's *MAY* compress indexes is a /// write-path option this version defers). pub fn operation_index(payload: Vec) -> Self { StagedChunk { kind: ChunkKind::OperationIndex, schema_version: SchemaVersion::V0, payload, } } } /// Context handed to a commit's manifest builder: the previous manifest, the /// [`ChunkRef`]s for the chunks this commit just wrote (in staging order), and /// the generation the new manifest will carry. pub struct CommitContext<'a> { /// The manifest active before this commit. pub previous_manifest: &'a Manifest, /// References to the chunks written by this commit, in staging order. pub new_chunks: &'a [ChunkRef], /// The generation the new manifest must declare (active + 1). pub generation: u64, /// The schema version the *previous* manifest was stamped with (G-minor, /// `spec/PLAN_GMINOR_SCHEMA_MINOR.md` §4, pin 6.2): new plumbing, not a /// read-through — [`CommitContext::previous_manifest`] is a `&Manifest`, /// and [`Manifest`] carries no schema-version field at all (it lives in /// the superblock, `Superblock::manifest_schema_version`). A build /// closure whose commit preserves the complete barrier content unchanged /// reads this to preserve the carried version exactly (pin 6.6), rather /// than recomputing it from scratch. pub previous_manifest_version: SchemaVersion, } /// An open bundle over a block store. pub struct Bundle { store: S, header: FixedHeader, active_slot: Slot, superblock: Superblock, manifest: Manifest, /// Next append offset. Always at end-of-file, so appends never overwrite a /// reachable chunk (Chapter 8 §"The Atomic Write Protocol", step 1). write_cursor: u64, read_only: bool, anomalies: Vec, } impl Bundle { /// Creates a brand-new bundle in `store` with the given file UUID and /// initial manifest (its generation is forced to 0). Writes the manifest /// chunk, then the header, the generation-0 superblock in slot A, and a /// zeroed (invalid) slot B; flushes at the manifest and at the superblock. /// /// A crash *during creation* may leave a half-formed file that [`Bundle::open`] /// rejects as corrupt — acceptable, since the file is not yet a bundle. The /// crash-safety guarantee is about *commits to an existing bundle*. /// /// Stamps the manifest chunk at the **baseline** [`Manifest::SCHEMA`] /// version. A manifest created here declares no canonical roots or /// blobs (enforced below), so it cannot yet name an edit barrier whose /// tag requires a higher epoch — see [`Bundle::create_versioned`] for a /// caller (e.g. a repack seeding a non-empty initial manifest through a /// different path) that must supply the version explicitly. pub fn create(store: S, file_uuid: FileUuid, manifest: Manifest) -> Result { Self::create_versioned(store, file_uuid, manifest, Manifest::SCHEMA) } /// As [`Bundle::create`], but the manifest chunk is stamped at the given /// `manifest_schema_version` rather than the baseline [`Manifest::SCHEMA`] /// (G-minor pin 6.1: "producers supply the aggregate manifest /// `SchemaVersion` explicitly"). `epiphany-bundle` itself never derives /// this value — it has no dependency on `epiphany-ops` or /// `epiphany-layout-ir` and cannot decode `prohibited_operation_kinds` — /// so the caller (a producer with the right dependencies) computes it. pub fn create_versioned( mut store: S, file_uuid: FileUuid, mut manifest: Manifest, manifest_schema_version: SchemaVersion, ) -> Result { manifest.generation = 0; manifest.manifest_id = manifest.derive_id(); // The manifest must be emittable (it would otherwise be rejected by // `open`): at least one declared profile. let active_profile = validate_emittable_manifest(&manifest)?; // A freshly created bundle writes only the manifest chunk, so it cannot // have any canonical roots or blobs yet: reject an initial manifest that // declares (necessarily dangling) operation roots, a canonical base, or // blob roots. if !manifest.operation_roots.is_empty() || manifest.canonical_base.is_some() || !manifest.blob_roots.is_empty() { return Err(BundleError::Decode(DecodeError::Malformed( "create() requires a manifest with no canonical roots or blobs", ))); } // Encode the manifest and enforce the reader's manifest-size limit *here* // (a writer must not emit a manifest its own `open` would reject as // oversize). Then normalize the in-memory copy to the canonical // (sorted/deduplicated) form by decoding the bytes we just produced, so // `bundle.manifest()` matches what a reopen would yield. let manifest_payload = manifest.encode(); enforce_limit(manifest_payload.len() as u64, MAX_MANIFEST_BYTES)?; let manifest = Manifest::decode(&manifest_payload)?; let manifest_hash = chunk_content_hash( ChunkKind::Manifest, manifest_schema_version, &manifest_payload, ); store.write_at(BODY_START, &manifest_payload)?; store.flush()?; let superblock = Superblock { generation: 0, manifest_offset: BODY_START, manifest_length: manifest_payload.len() as u64, manifest_hash, manifest_schema_version, reduction_algorithm_version: reduction_version_for(&manifest), profile_id: active_profile.profile_id, commit_state: CommitState::Committed, commit_timestamp: WallClockTime(0), }; let header = FixedHeader::new(file_uuid); store.write_at(0, &header.encode())?; store.write_at(SLOT_A_OFFSET, &superblock.encode())?; // Slot B is explicitly zeroed so it is invalid (bad magic) until the // first commit writes a real superblock there. store.write_at(SLOT_B_OFFSET, &[0u8; SUPERBLOCK_LEN as usize])?; store.flush()?; let write_cursor = store.len(); // A required unknown extension, or a non-editable (`ReadOnly`) active // profile, makes even a freshly created bundle read-only. let read_only = manifest_forces_read_only(&manifest) || !profile_is_editable(&active_profile); Ok(Bundle { store, header, active_slot: Slot::A, superblock, manifest, write_cursor, read_only, anomalies: Vec::new(), }) } /// Opens an existing bundle: the cold-open procedure (Chapter 8 /// §"Cold Open Procedure"). Verifies the header (magic, CRC), selects the /// active superblock (validating each slot's magic, CRC, commit state, and /// manifest hash), and reads + verifies the manifest. A structural anomaly /// (generation gap, divergent same-generation slots, a non-committed slot) /// opens the bundle **read-only** and is recorded in [`Bundle::anomalies`]. pub fn open(store: S) -> Result { // 1. Header. let header_bytes = read_vec(&store, 0, crate::header::HEADER_LEN)?; let header = FixedHeader::decode(&header_bytes)?; // 2. Both slots. let slot_a_bytes = read_vec(&store, SLOT_A_OFFSET, SUPERBLOCK_LEN)?; let slot_b_bytes = read_vec(&store, SLOT_B_OFFSET, SUPERBLOCK_LEN)?; let mut anomalies = Vec::new(); // 3. Parse + manifest-hash-verify each slot. A slot survives only if it // is a valid, committed superblock whose manifest chunk verifies. let a = verified_slot(&store, &slot_a_bytes, &mut anomalies); let b = verified_slot(&store, &slot_b_bytes, &mut anomalies); // 4–6. Apply the selection rule. A *selection-level* anomaly (a // generation gap > 1, or divergent slots at the same generation) forces // read-only recovery — these are the cases Chapter 8 says MUSTNOT be // treated as normal. A non-committed slot (recorded by `verified_slot`) // is ordinary fallback per the atomic-write protocol: it is surfaced but // does not, on its own, block editing. let selection = select_active(a, b)?; let mut read_only = selection.anomaly.is_some(); if let Some(anomaly) = selection.anomaly.clone() { anomalies.push(anomaly); } let superblock = selection.superblock; // A manifest at an unsupported schema major cannot be interpreted as // canonical state (Chapter 8 §"Schema Versioning"). The manifest stays // at its own schema major across the schema-major-1 bump (its body is // carried opaquely and never grows a v1 layout here), so this gate is // exact to `Manifest::SCHEMA.major` — it must not admit a manifest major // that has no defined wire form. if superblock.manifest_schema_version.major != Manifest::SCHEMA.major { return Err(BundleError::UnsupportedSchemaVersion { version: superblock.manifest_schema_version, }); } // Read and decode the manifest the selected superblock points at. The // placement (in-body, within the size limit) and hash were already // checked by `verified_slot`; `Manifest::decode` additionally rejects // non-canonical bytes. let manifest_payload = read_manifest_payload(&store, &superblock)?; let manifest = Manifest::decode(&manifest_payload)?; // The manifest's self-declared generation must match the superblock that // referenced it (a mismatch is structural corruption). if manifest.generation != superblock.generation { return Err(BundleError::GenerationMismatch { superblock: superblock.generation, manifest: manifest.generation, }); } // Every bundle MUST declare at least one profile, with distinct ids // (Chapter 8 §"Format Profiles"). if manifest.profile_declarations.is_empty() { return Err(BundleError::Decode(DecodeError::Malformed( "manifest declares no conformance profile", ))); } if !profile_ids_distinct(&manifest) { return Err(BundleError::Decode(DecodeError::Malformed( "manifest declares the same profile id more than once", ))); } // The active superblock's profile must be one the manifest declares. let active_profile = manifest .profile_declarations .iter() .find(|p| p.profile_id == superblock.profile_id) .copied(); let active_profile = match active_profile { Some(p) => p, None => { return Err(BundleError::Decode(DecodeError::Malformed( "superblock profile is not declared by the manifest", ))) } }; // If the active profile is not understood (a `Custom` registry profile, // a future major, or a block bound beyond the reader's limit), open // read-only and surface it; if it is understood but not editable // (`ReadOnly`), open read-only silently. (Chapter 8 §"Format Profiles": // a reader edits only under a profile it supports.) if !profile_is_understood(&active_profile) { read_only = true; anomalies.push(IntegrityAnomaly::UnsupportedProfile); } else if !profile_is_editable(&active_profile) { read_only = true; } // A canonical base is usable only if its reduction-algorithm version // matches the active superblock's *and* its profile is one the manifest // declares (Chapter 8 §"Canonical Document Identity"). if let Some(base) = &manifest.canonical_base { if base.reduction_algorithm_version != superblock.reduction_algorithm_version { return Err(BundleError::Decode(DecodeError::Malformed( "canonical base reduction-algorithm version disagrees with the superblock", ))); } if !manifest .profile_declarations .iter() .any(|p| p.profile_id == base.profile_id) { return Err(BundleError::Decode(DecodeError::Malformed( "canonical base profile is not declared by the manifest", ))); } // Format-epoch matrix (`CONTRACT_FORMAT_EPOCH_MAJOR1.md` pin 3, // rows 2 and 5i). Corruption precedence runs first (both checks // above): only once the base is structurally sound do we ask // whether this epoch may open it at all. A legacy (major-0) // container can never open with a base — the epoch is not // retroactive (row 2, permanent). A major-1 container is the // right epoch, but until P13-S27 lands there is no // reduction-authority capability to validate it against (row 5i, // interim — TEMPORARY, removed by P13-S27, pin 3a). return Err(match header.epoch { FormatEpoch::Legacy => BundleError::LegacyBundleHasCanonicalBase, FormatEpoch::Current => BundleError::ReductionAuthorityUnavailable, }); } // An unknown *required* extension forces read-only (Chapter 8 §"Behavior // Under Unknown Extensions"). if manifest_forces_read_only(&manifest) { read_only = true; anomalies.push(IntegrityAnomaly::UnknownRequiredExtension); } // A canonical operation root at a schema major above this reader's // accept-set forces read-only preservation (Binary Format companion // §"Schema Major 1"): the reader still reads the canonical base and // manifest (both major 0) but refuses to author against op history it // cannot interpret. A cheap manifest-metadata scan; the beyond-accept // blocks are never read (a lazy read would hit the accept-set gate). if let Some(major) = unsupported_operation_root_major(&manifest) { read_only = true; anomalies.push(IntegrityAnomaly::UnsupportedCanonicalChunkMajor { schema_major: major, }); } // The canonical base is role-bound to major 0 (see // `mis_stamped_canonical_base`). if let Some(major) = mis_stamped_canonical_base(&manifest) { read_only = true; anomalies.push(IntegrityAnomaly::UnsupportedCanonicalChunkMajor { schema_major: major, }); } let write_cursor = store.len(); Ok(Bundle { store, header, active_slot: selection.slot, superblock, manifest, write_cursor, read_only, anomalies, }) } /// The active manifest. pub fn manifest(&self) -> &Manifest { &self.manifest } /// The active generation. pub fn generation(&self) -> u64 { self.superblock.generation } /// The fixed header. pub fn header(&self) -> &FixedHeader { &self.header } /// The active superblock. pub fn superblock(&self) -> &Superblock { &self.superblock } /// Which slot is currently active. pub fn active_slot(&self) -> Slot { self.active_slot } /// The physical-bundle UUID. pub fn file_uuid(&self) -> FileUuid { self.header.file_uuid } /// Whether the bundle is open read-only (an integrity anomaly was detected, /// or an unknown required extension). Commits are refused. pub fn is_read_only(&self) -> bool { self.read_only } /// Structural anomalies detected at open (empty for a normal bundle). pub fn anomalies(&self) -> &[IntegrityAnomaly] { &self.anomalies } /// The underlying store. pub fn store(&self) -> &S { &self.store } /// Consumes the bundle, returning its store (e.g. to inspect the bytes). pub fn into_store(self) -> S { self.store } /// Reads and fully verifies a chunk against its reference: that it lies in /// the body (not the prelude), that its schema major is supported, its /// declared length, its BLAKE3 content hash, and the `id == hash` redundancy /// the spec keeps (Chapter 8 §"Chunks"). A canonical chunk whose hash fails /// is hard corruption ([`BundleError::ChunkHashMismatch`]). pub fn read_chunk(&self, r: &ChunkRef) -> Result, BundleError> { read_and_verify_chunk(&self.store, r) } /// Reads and verifies a blob (Chapter 8 §"Blobs"): the same checks as /// [`Bundle::read_chunk`], but addressed via the bare `MUSCBLOB` hash. A /// resource-limit check honors the blob's `declared_max_uncompressed_length`. pub fn read_blob(&self, b: &BlobRef) -> Result, BundleError> { read_and_verify_blob(&self.store, b, MAX_BLOB_BYTES) } /// Reads a chunk and splits it into its opaque operation-envelope byte /// strings (Chapter 8 §"Operation Envelope Blocks"). Rejects a reference of /// the wrong kind, and a block whose uncompressed size exceeds the active /// profile's maximum block size. The bundle does not interpret the envelopes. pub fn read_operation_block(&self, r: &ChunkRef) -> Result>, BundleError> { if r.kind != ChunkKind::OperationEnvelopeBlock { return Err(BundleError::Decode(DecodeError::Malformed( "chunk reference is not an operation-envelope block", ))); } if r.uncompressed_length > self.max_block_size() { return Err(BundleError::Decode(DecodeError::Malformed( "operation-envelope block exceeds the active profile's maximum size", ))); } let payload = self.read_chunk(r)?; block::decode_block(&payload).map_err(BundleError::Decode) } /// Reads and decodes an operation-index chunk (Chapter 8 §"The Operation /// Index"). Rejects a reference of the wrong kind; the read itself is /// bounded by the reader's [`MAX_CHUNK_BYTES`] policy and hash-verified /// like any chunk read. /// /// **The index is not canonical**: a failure here — corrupt bytes, a /// malformed payload, even an I/O error on the index region — must NOT be /// treated as bundle corruption. The spec's discipline is *reject and /// rebuild from blocks*; [`Bundle::usable_operation_index`] packages it /// (including the staleness check). Call this directly only when the /// underlying failure itself is wanted, e.g. for diagnostics. pub fn read_operation_index(&self, r: &ChunkRef) -> Result { if r.kind != ChunkKind::OperationIndex { return Err(BundleError::Decode(DecodeError::Malformed( "chunk reference is not an operation index", ))); } let payload = self.read_chunk(r)?; OperationIndex::decode(&payload).map_err(BundleError::Decode) } /// The manifest's operation index, if — and only if — it is *usable*: /// declared, readable, hash-intact, well-formed, and covering exactly the /// manifest's current `operation_roots` ([`OperationIndex::covers`]). /// `None` on **any** defect: absent, stale (its block set differs from the /// operation roots), corrupt, malformed, or unreadable. /// /// `None` always means "rebuild by scanning all blocks", never "the bundle /// is corrupt": the operation index is an acceleration structure, not /// canonical (Chapter 8 §"The Operation Index"), and failed verification /// of a non-canonical chunk MUST NOT be surfaced as bundle corruption /// (Chapter 8 §"Canonical and Non-Canonical Manifest Roots") — the reader /// discards the index and rebuilds it from the blocks. pub fn usable_operation_index(&self) -> Option { let root = self.manifest.operation_index_root.as_ref()?; let index = self.read_operation_index(root).ok()?; index .covers(&self.manifest.operation_roots) .then_some(index) } /// The active profile's maximum uncompressed operation-block size /// (Chapter 8 §"Operation Envelope Blocks"). The *active* profile is the one /// the selected superblock names — a bundle opened under `Lite` must read /// under `Lite`'s limits, not the canonical-first profile's. Falls back to /// the canonical-first declaration, then the 64 MiB default. fn max_block_size(&self) -> u64 { self.active_profile() .or_else(|| self.manifest.canonical_first_profile()) .map(|p| p.constraints.max_uncompressed_block_size) .unwrap_or(block::MAX_BLOCK_DEFAULT) } /// The profile declaration the active superblock names, if the manifest /// declares it. fn active_profile(&self) -> Option { self.manifest .profile_declarations .iter() .find(|p| p.profile_id == self.superblock.profile_id) .copied() } /// Verifies every canonical chunk reachable from the manifest — the operation /// blocks, the canonical base (its root chunk, with the snapshot's restated /// hash cross-checked against the root), and every declared blob — is present /// and intact. The crash-recovery and cold-open paths call this to honor the /// spec rule that failed verification of a *canonical* chunk is hard /// corruption. (The bundle cannot tell *which* blobs are canonical without /// interpreting operations, so it conservatively verifies all of them.) pub fn verify_canonical_chunks(&self) -> Result<(), BundleError> { for r in &self.manifest.operation_roots { read_and_verify_chunk(&self.store, r)?; } if let Some(base) = &self.manifest.canonical_base { read_and_verify_chunk(&self.store, &base.root)?; if base.hash != base.root.hash { return Err(BundleError::ChunkHashMismatch { expected: base.root.hash, actual: base.hash, }); } } for b in &self.manifest.blob_roots { self.read_blob(b)?; } Ok(()) } /// Commits a new generation via the seven-step atomic write protocol. /// /// `new_chunks` are written (and flushed) first; the `build` closure then /// assembles the new manifest from the resulting [`ChunkRef`]s and the /// previous manifest; the manifest is written (and flushed); finally a new /// superblock at `generation + 1` is written to the inactive slot and /// flushed — the commit point. On success the in-memory state advances to /// the new generation. /// /// On a store error *before* the commit-point flush, the active superblock is /// untouched, so the bundle is unchanged on disk except for unreachable /// appended bytes. On an error *at* the commit-point flush the durable result /// is indeterminate (the new superblock may or may not have landed); the /// bundle is poisoned read-only and the caller must reopen from storage. /// /// Stamps the new manifest chunk at the **baseline** [`Manifest::SCHEMA`] /// version. See [`Bundle::commit_versioned`] for a caller that must /// supply a non-baseline aggregate (e.g. because the manifest the /// closure builds names an edit barrier prohibiting a post-baseline /// `OperationKindTag`). pub fn commit( &mut self, new_chunks: &[StagedChunk], build: impl FnOnce(&CommitContext) -> Manifest, ) -> Result<(), BundleError> { self.commit_versioned(new_chunks, Manifest::SCHEMA, build) } /// As [`Bundle::commit`], but the new manifest chunk is stamped at the /// given `manifest_schema_version` rather than the baseline /// [`Manifest::SCHEMA`] (G-minor pin 6.1). The build closure reads /// [`CommitContext::previous_manifest_version`] if it needs to decide /// whether the previous aggregate is still exact (pin 6.6: an ordinary /// repack preserving the complete barrier content preserves the carried /// version exactly). pub fn commit_versioned( &mut self, new_chunks: &[StagedChunk], manifest_schema_version: SchemaVersion, build: impl FnOnce(&CommitContext) -> Manifest, ) -> Result<(), BundleError> { if self.read_only { return Err(BundleError::ReadOnly); } let next_generation = self .superblock .generation .checked_add(1) .ok_or(BundleError::GenerationExhausted)?; let inactive = self.active_slot.other(); let mut cursor = self.write_cursor; // Step 1: write all new chunks outside the prelude and reachable chunks // (we always append at EOF, so the "MUSTNOT overwrite a reachable chunk" // rule holds by construction). Content-addressed dedup: a staged chunk // whose content hash already exists — referenced by the current manifest // or written earlier in this same commit — reuses that storage instead // of re-appending (Chapter 8: duplicate content shares storage). let mut known: BTreeMap = self .manifest .referenced_chunk_refs() .into_iter() .map(|r| (r.id, r)) .collect(); // Index blobs too (keyed by their bare content hash), so re-staging a // payload already retained only as a `BlobRef` reuses its storage. for b in &self.manifest.blob_roots { known.entry(ChunkId(b.hash)).or_insert(ChunkRef { id: ChunkId(b.hash), kind: ChunkKind::Blob, schema_version: SchemaVersion::V0, offset: b.offset, compressed_length: b.compressed_length, uncompressed_length: b.uncompressed_length, compression: b.compression, hash: b.hash, }); } let mut new_refs = Vec::with_capacity(new_chunks.len()); for staged in new_chunks { let id = chunk_id(staged.kind, staged.schema_version, &staged.payload); let r = if let Some(existing) = known.get(&id) { *existing } else { let r = append_chunk( &mut self.store, &mut cursor, staged.kind, staged.schema_version, &staged.payload, )?; known.insert(id, r); r }; new_refs.push(r); } // Step 2: durable flush. self.store.flush()?; // Build the new manifest from the previous one and the new chunk refs. let previous = self.manifest.clone(); let mut manifest = build(&CommitContext { previous_manifest: &previous, new_chunks: &new_refs, generation: next_generation, previous_manifest_version: self.superblock.manifest_schema_version, }); // Extension-root preservation (Chapter 8 §"Behavior Under Unknown // Extensions"). The bundle's job is preservation: an extension-unaware // commit closure must not silently drop unknown extensions and their // `preserved_chunk_roots` (which would orphan those chunks). Carry // forward every prior extension declaration the closure did not itself // re-declare; an extension-*aware* writer that re-declares its own // `extension_id` keeps full control of that declaration. (The manifest // encoder sorts/dedups `extension_declarations`, so append order does not // affect the canonical form.) let redeclared: std::collections::BTreeSet = manifest .extension_declarations .iter() .map(|e| e.extension_id) .collect(); for prior in &previous.extension_declarations { if !redeclared.contains(&prior.extension_id) { manifest.extension_declarations.push(prior.clone()); } } manifest.generation = next_generation; manifest.manifest_id = manifest.derive_id(); // The new manifest must be emittable (else `open` would reject it): at // least one declared profile. let active_profile = validate_emittable_manifest(&manifest)?; // Format-epoch matrix (`CONTRACT_FORMAT_EPOCH_MAJOR1.md` pin 3, rows 3 // and 6i): a commit may never introduce (or replace) a canonical // base. `self.manifest.canonical_base` is always `None` here — `open` // and `create` both already refuse ever producing a live bundle whose // base is `Some`, so any `Some` reaching this point is necessarily a // fresh introduction. Row 3 — a legacy (major-0) container can never // become base-bearing in place, because its header can never say the // base was validated (permanent; the epoch is not inheritable). Row // 6i — a major-1 container is the right epoch, but until P13-S27 // lands there is no reduction-authority capability to validate a // newly introduced base against (interim — TEMPORARY, removed by // P13-S27, pin 3a). Placed after `validate_emittable_manifest` (which // already rejects a base whose profile is undeclared) so that a // structurally malformed base is still reported as malformed, not // masked by this categorical epoch refusal. if manifest.canonical_base.is_some() { return Err(match self.header.epoch { FormatEpoch::Legacy => BundleError::LegacyBaseIntroductionRejected, FormatEpoch::Current => BundleError::ReductionAuthorityUnavailable, }); } // Before publishing, validate that every canonical root the new manifest // declares actually resolves to a present, hash-intact chunk of the right // kind and shape (the new chunks are written and flushed; retained roots // are already in the body). This refuses a builder closure that produced // dangling, wrong-kind, or mismatched roots — an abort here leaves the // active superblock untouched, so the bundle stays at the old generation. validate_canonical_roots(&self.store, &manifest, &active_profile)?; // Step 3: write the new (uncompressed) manifest chunk. Enforce the // reader's manifest-size limit before publishing (else the new slot would // reopen as oversize), and normalize the in-memory copy to the canonical // form by decoding the bytes we produce — so `bundle.manifest()` matches // a reopen (e.g. duplicate roots are already collapsed). let manifest_payload = manifest.encode(); enforce_limit(manifest_payload.len() as u64, MAX_MANIFEST_BYTES)?; let manifest = Manifest::decode(&manifest_payload)?; let manifest_hash = chunk_content_hash( ChunkKind::Manifest, manifest_schema_version, &manifest_payload, ); let manifest_offset = cursor; self.store.write_at(manifest_offset, &manifest_payload)?; cursor += manifest_payload.len() as u64; // Step 4: durable flush. self.store.flush()?; // Step 5: compute the new superblock. let superblock = Superblock { generation: next_generation, manifest_offset, manifest_length: manifest_payload.len() as u64, manifest_hash, manifest_schema_version, reduction_algorithm_version: reduction_version_for(&manifest), profile_id: active_profile.profile_id, commit_state: CommitState::Committed, commit_timestamp: WallClockTime(0), }; // Step 6: write it to the currently-inactive slot. self.store .write_at(inactive.offset(), &superblock.encode())?; // Step 7: durable flush — the commit point. An error here is // *indeterminate*: the new superblock may or may not have reached durable // storage, so the on-disk active generation is now unknown and the // in-memory state cannot be trusted. Poison the bundle (read-only) so no // further commit runs against stale state and risks overwriting the slot // that may have just become active; the caller must reopen from storage. if let Err(e) = self.store.flush() { self.read_only = true; return Err(e.into()); } // The commit is durable; advance the in-memory state. If the new manifest // introduces a required unknown extension or a non-editable active // profile, the bundle becomes read-only immediately (not only on the next // open). self.active_slot = inactive; self.superblock = superblock; self.read_only = manifest_forces_read_only(&manifest) || !profile_is_editable(&active_profile); // A commit that publishes a canonical operation root beyond this reader's // accept-set (a forward-compat write) makes the *live* bundle read-only // at once — mirroring the open-time scan — so no further commit runs // against canonical history this reader can no longer parse. if let Some(major) = unsupported_operation_root_major(&manifest) { self.read_only = true; self.anomalies .push(IntegrityAnomaly::UnsupportedCanonicalChunkMajor { schema_major: major, }); } if let Some(major) = mis_stamped_canonical_base(&manifest) { self.read_only = true; self.anomalies .push(IntegrityAnomaly::UnsupportedCanonicalChunkMajor { schema_major: major, }); } self.manifest = manifest; self.write_cursor = cursor; Ok(()) } } impl Bundle { /// The current in-memory bundle image (only available for the in-memory /// store). Useful for snapshotting a bundle's bytes between commits. pub fn image(&self) -> &[u8] { self.store.as_bytes() } } /// Whether a manifest is well-formed enough to *emit* — a writer must not emit a /// manifest its own `open` would reject. Chapter 8 §"Format Profiles" requires at /// least one declared profile, and §"Canonical Document Identity" requires a /// canonical base's profile to be one the manifest declares. fn validate_emittable_manifest(manifest: &Manifest) -> Result { if manifest.profile_declarations.is_empty() { return Err(BundleError::Decode(DecodeError::Malformed( "manifest declares no conformance profile", ))); } if !profile_ids_distinct(manifest) { return Err(BundleError::Decode(DecodeError::Malformed( "manifest declares the same profile id more than once", ))); } // The active profile this manifest would be emitted under must be one this // implementation *understands* (can interpret and honor). It need not be // editable: a bundle may legitimately be produced under `ReadOnly` (it just // opens read-only). This refuses emitting only under an unknown/`Custom`, // wrong-major, or oversize-block profile — there is no understood profile to // operate under. let active_profile = active_profile_for_emit(manifest).ok_or(BundleError::Decode( DecodeError::Malformed("no declared profile is supported by this implementation"), ))?; if let Some(base) = &manifest.canonical_base { if !manifest .profile_declarations .iter() .any(|p| p.profile_id == base.profile_id) { return Err(BundleError::Decode(DecodeError::Malformed( "canonical base profile is not declared by the manifest", ))); } } Ok(active_profile) } /// Whether a manifest forces read-only mode: it declares a `required` extension /// this implementation does not understand (v0 understands none, so any /// `required` extension qualifies — Chapter 8 §"Behavior Under Unknown /// Extensions"). fn manifest_forces_read_only(manifest: &Manifest) -> bool { manifest.extension_declarations.iter().any(|e| e.required) } /// The schema major of a declared **canonical operation root** that is beyond /// this reader's accept-set for the op-block role, if any (Binary Format /// companion §"Schema Major 1", "Canonical chunks — parse or open read-only"). /// Such a root forces read-only preservation: this reader cannot parse the newer /// op bytes, so it must not author against op history it cannot interpret. Both /// `open` (at load) and `commit` (a builder that publishes such a root, e.g. a /// forward-compat write) consult this so the in-memory bundle goes read-only /// immediately, not only on the next reopen. fn unsupported_operation_root_major(manifest: &Manifest) -> Option { manifest .operation_roots .iter() .map(|r| r.schema_version.major) .find(|&m| m > max_supported_major(ChunkKind::OperationEnvelopeBlock)) } /// The canonical base MUST stay schema major 0 across the data-model majors /// (Binary Format §Schema Major 1 / §Schema Major 2: the `MaterializedState` /// embeds none of the filled values, and re-stamping byte-identical content /// churns its content address). With the Snapshot *kind* now admitting the /// major-2 acceleration form, this per-ROLE check keeps the base constraint: /// a base ref stamped above major 0 forces read-only preservation, exactly /// like a beyond-accept-set op root. fn mis_stamped_canonical_base(manifest: &Manifest) -> Option { manifest .canonical_base .as_ref() .map(|b| b.root.schema_version.major) .filter(|&m| m > 0) } /// Whether this implementation *understands* a profile (can interpret and honor /// its constraints): a built-in `ProfileId` (not a `Custom` registry profile), /// a supported major version, and a block bound within the reader's hard chunk /// limit (a profile demanding larger blocks than the reader can allocate is not /// honorable — Chapter 8 §"Operation Envelope Blocks" / §"Format Profiles"). fn profile_is_understood(decl: &ProfileDeclaration) -> bool { // A profile major change is non-backward-compatible (like a schema major), so // the major must match exactly; minor/patch are accepted. matches!( decl.profile_id, ProfileId::Full | ProfileId::ReadOnly | ProfileId::Lite ) && decl.version.major == SUPPORTED_PROFILE_MAJOR && decl.constraints.max_uncompressed_block_size <= MAX_CHUNK_BYTES } /// Whether a profile is *editable*: understood, and not the `ReadOnly` profile. /// A `ReadOnly`-profile bundle opens read-only; v0 does not auto-upgrade it to a /// writable profile (the spec's SHOULD-upgrade-on-edit is deferred). fn profile_is_editable(decl: &ProfileDeclaration) -> bool { profile_is_understood(decl) && matches!(decl.profile_id, ProfileId::Full | ProfileId::Lite) } /// Whether a manifest's profile declarations have distinct `ProfileId`s. A /// duplicate id (even at a different version/constraints) is ambiguous about /// which declaration governs and is rejected. fn profile_ids_distinct(manifest: &Manifest) -> bool { let mut ids: Vec = manifest .profile_declarations .iter() .map(|p| p.profile_id) .collect(); ids.sort(); let len = ids.len(); ids.dedup(); ids.len() == len } /// The profile a freshly emitted superblock should name as active: prefer the /// canonical-first *editable* profile, so the bundle is editable whenever it /// declares an editable profile (e.g. `[ReadOnly, Lite]` is emitted under /// `Lite`); otherwise the canonical-first merely *understood* profile (e.g. a /// sole `ReadOnly` yields a read-only bundle). `None` if no declared profile is /// understood — `validate_emittable_manifest` rejects that before this is used. fn active_profile_for_emit(manifest: &Manifest) -> Option { let profiles = manifest.canonical_profiles(); profiles .iter() .find(|p| profile_is_editable(p)) .or_else(|| profiles.iter().find(|p| profile_is_understood(p))) .copied() } /// The reduction-algorithm version a superblock should carry: the canonical /// base's, if a base is present (only a base records a reduction); otherwise the /// default (no base means no reduced base state at this generation). fn reduction_version_for(manifest: &Manifest) -> ReductionAlgorithmVersion { manifest .canonical_base .as_ref() .map(|b| b.reduction_algorithm_version) .unwrap_or_default() } /// Reads and fully verifies a chunk against its reference — the shared core of /// [`Bundle::read_chunk`] and the commit-time canonical-root validation. Checks /// compression support (decompressing zstd payloads), body-placement, /// schema-major support, declared length, the content hash, and the /// `id == hash` redundancy (Chapter 8 §"Chunks"). fn read_and_verify_chunk(store: &dyn BlockStore, r: &ChunkRef) -> Result, BundleError> { read_and_verify_chunk_impl(store, r, true) } /// [`read_and_verify_chunk`] with the schema-major **accept-set** check made /// optional. The accept-set is a *reader-capability* gate, not a structural /// property: a bundle may legitimately carry a canonical root written by a /// newer writer at a major beyond this reader's accept-set. Commit-time /// canonical-root validation therefore checks a root's *structure* (resolves, /// hash-intact, right kind, decodes) with `enforce_accept_set = false` — it must /// not refuse to publish a root it merely cannot itself parse — while every read /// path enforces the accept-set (a beyond-accept-set canonical root instead /// opens the bundle read-only at `open`). fn read_and_verify_chunk_impl( store: &dyn BlockStore, r: &ChunkRef, enforce_accept_set: bool, ) -> Result, BundleError> { // The manifest chunk is mandatorily uncompressed in this format version // (Chapter 8 §"Manifest Encoding"): a compressed manifest reference is // rejected outright, before any bytes are read. if r.kind == ChunkKind::Manifest && r.compression != CompressionAlgorithm::None { return Err(BundleError::CompressedManifest); } if let CompressionAlgorithm::Reserved(_) = r.compression { return Err(BundleError::UnsupportedCompression); } // A chunk reference must point into the body, never the fixed prelude. if r.offset < BODY_START { return Err(BundleError::ChunkOutOfBounds { offset: r.offset, length: r.compressed_length, file_len: store.len(), }); } // A chunk at a schema major this reader cannot parse for its role. The // accept-set is `[0, max_supported_major(kind)]`: the op-block and // snapshot roles admit major 2, every other role stays exact-0 until its // versioned path lands (Binary Format companion §"Schema Major 2"). A // chunk above its role's max reaches this only on a direct read — a // canonical root beyond the accept-set opens the bundle read-only at // `open` instead, before any such read (see the operation-root and // canonical-base scans there). Commit-time structural validation skips // this gate (a newer writer's higher-major root is still structurally // valid). if enforce_accept_set && r.schema_version.major > max_supported_major(r.kind) { return Err(BundleError::UnsupportedSchemaVersion { version: r.schema_version, }); } // Bound both allocations by the reader's policy before touching a length. enforce_limit(r.compressed_length, MAX_CHUNK_BYTES)?; enforce_limit(r.uncompressed_length, MAX_CHUNK_BYTES)?; let stored = read_chunk_bytes(store, r.offset, r.compressed_length)?; let payload = decode_stored_payload(stored, r.compression, r.uncompressed_length)?; let actual = content_hash_for(r.kind, r.schema_version, &payload); if actual != r.hash { return Err(BundleError::ChunkHashMismatch { expected: r.hash, actual, }); } if r.id.content_hash() != r.hash { return Err(BundleError::ChunkHashMismatch { expected: r.hash, actual: r.id.content_hash(), }); } Ok(payload) } /// Reads and fully verifies a blob against its reference (the shared core of /// [`Bundle::read_blob`] and the commit-time blob validation): compression /// support, the reader resource limit (`min(max_bytes, declared_max)`), /// body-placement, declared length, and the bare-`MUSCBLOB` content hash with /// the `blob_id == hash` redundancy. fn read_and_verify_blob( store: &dyn BlockStore, b: &BlobRef, max_bytes: u64, ) -> Result, BundleError> { if let CompressionAlgorithm::Reserved(_) = b.compression { return Err(BundleError::UnsupportedCompression); } let limit = b .declared_max_uncompressed_length .unwrap_or(u64::MAX) .min(max_bytes); // Chapter 8 §"Blobs": the declared uncompressed length is checked against // the reader's policy *before decompression begins* (and before any // allocation keyed on it). enforce_limit(b.uncompressed_length, limit)?; enforce_limit(b.compressed_length, max_bytes)?; if b.offset < BODY_START { return Err(BundleError::ChunkOutOfBounds { offset: b.offset, length: b.compressed_length, file_len: store.len(), }); } let stored = read_chunk_bytes(store, b.offset, b.compressed_length)?; let payload = decode_stored_payload(stored, b.compression, b.uncompressed_length)?; let actual = BlobId::of_payload(&payload).0; if actual != b.hash || b.blob_id.0 != b.hash { return Err(BundleError::ChunkHashMismatch { expected: b.hash, actual, }); } Ok(payload) } /// Errors if `length` exceeds `limit`, before any allocation keyed on it. fn enforce_limit(length: u64, limit: u64) -> Result<(), BundleError> { if length > limit { Err(BundleError::ResourceLimitExceeded { length, limit }) } else { Ok(()) } } /// Recovers a chunk's uncompressed payload from its stored (possibly /// compressed) bytes, verifying it is *exactly* `declared_len` bytes long /// (Chapter 8 §"Compression" / §"Chunks": a decompressed size that disagrees /// with the declared `uncompressed_length` is corruption). The caller has /// already validated `declared_len` against its resource-limit policy, so /// every allocation here is bounded. Content hashing happens strictly *after* /// this step, over the uncompressed bytes — compression is `ChunkRef` /// metadata, never part of content identity. fn decode_stored_payload( stored: Vec, compression: CompressionAlgorithm, declared_len: u64, ) -> Result, BundleError> { match compression { CompressionAlgorithm::None => { if stored.len() as u64 != declared_len { return Err(BundleError::ChunkLengthMismatch { expected: declared_len, actual: stored.len() as u64, }); } Ok(stored) } // Reading zstd at any level is a conformance MUST (Chapter 8 // §"Compression"); the declared level byte is advisory metadata the // decoder does not need. CompressionAlgorithm::Zstd { .. } => decompress_zstd(&stored, declared_len), CompressionAlgorithm::Reserved(_) => Err(BundleError::UnsupportedCompression), } } /// Decompresses a zstd frame sequence into a buffer sized *exactly* by the /// declared uncompressed length, so a hostile stream can never allocate past /// the (already limit-checked) declaration: /// /// * a stream that would exceed `declared_len` hits libzstd's /// destination-full error → [`BundleError::Decompression`]; /// * a stream that ends short of `declared_len` → /// [`BundleError::ChunkLengthMismatch`]; /// * a truncated or otherwise malformed stream (including trailing garbage /// after the final frame) → [`BundleError::Decompression`]. /// /// No path panics or allocates beyond `declared_len` plus libzstd's own /// bounded decoding context (whose window is capped internally). fn decompress_zstd(stored: &[u8], declared_len: u64) -> Result, BundleError> { let declared = usize::try_from(declared_len).map_err(|_| { // Unreachable on 64-bit targets; on narrower ones an unaddressable // declaration is a resource-limit refusal, not a wrap. BundleError::ResourceLimitExceeded { length: declared_len, limit: usize::MAX as u64, } })?; let mut payload = vec![0u8; declared]; match zstd::bulk::decompress_to_buffer(stored, payload.as_mut_slice()) { Ok(n) if n as u64 == declared_len => Ok(payload), Ok(n) => Err(BundleError::ChunkLengthMismatch { expected: declared_len, actual: n as u64, }), Err(e) => Err(BundleError::Decompression(e)), } } /// Validates that every canonical root a manifest declares resolves to a /// present, hash-intact chunk of the *right kind and shape* before a commit /// publishes it — so a bundle that opens normally can never carry a dangling, /// mis-roled, or malformed canonical root: /// /// * each operation root is an `OperationEnvelopeBlock`, within the active /// profile's maximum block size, whose payload actually decodes; /// * the canonical base's root is a `Snapshot`, with its restated hash matching; /// * every declared blob root resolves and verifies. fn validate_canonical_roots( store: &dyn BlockStore, manifest: &Manifest, active_profile: &ProfileDeclaration, ) -> Result<(), BundleError> { let max_block = active_profile.constraints.max_uncompressed_block_size; for r in &manifest.operation_roots { if r.kind != ChunkKind::OperationEnvelopeBlock { return Err(BundleError::Decode(DecodeError::Malformed( "operation root is not an operation-envelope block", ))); } if r.uncompressed_length > max_block { return Err(BundleError::Decode(DecodeError::Malformed( "operation block exceeds the active profile's maximum size", ))); } // Structural verification only (`enforce_accept_set = false`): a commit // may publish an op block at a major beyond this reader's accept-set (a // newer writer's v1+ block). The accept-set is enforced on read, and a // beyond-accept-set canonical root opens the bundle read-only at `open`. let payload = read_and_verify_chunk_impl(store, r, false)?; // The block payload must be a well-formed envelope sequence. block::decode_block(&payload).map_err(BundleError::Decode)?; } if let Some(base) = &manifest.canonical_base { if base.root.kind != ChunkKind::Snapshot { return Err(BundleError::Decode(DecodeError::Malformed( "canonical base root is not a snapshot chunk", ))); } read_and_verify_chunk(store, &base.root)?; if base.hash != base.root.hash { return Err(BundleError::ChunkHashMismatch { expected: base.root.hash, actual: base.hash, }); } } for b in &manifest.blob_roots { if !crate::manifest::valid_media_type(&b.media_type) { return Err(BundleError::Decode(DecodeError::Malformed( "blob media type is not a valid RFC 6838 type/subtype", ))); } read_and_verify_blob(store, b, MAX_BLOB_BYTES)?; } Ok(()) } /// Reads the manifest payload a superblock points at, enforcing that it lies in /// the body — Chapter 8 fixes the prelude layout, so a "manifest" overlapping a /// header or superblock slot is foreign — and within the reader's manifest size /// limit, before allocating. /// /// The manifest chunk is mandatorily *uncompressed* in this format version /// (Chapter 8 §"Manifest Encoding"): the superblock deliberately carries no /// compression field, so the stored bytes ARE the payload. A bundle whose /// manifest bytes are compressed anyway therefore fails downstream as /// malformed: hash verification (over the uncompressed preimage) rejects the /// slot, and even a colluding hash-over-compressed-bytes cannot survive /// `Manifest::decode`. fn read_manifest_payload(store: &dyn BlockStore, sb: &Superblock) -> Result, BundleError> { if sb.manifest_offset < BODY_START { return Err(BundleError::ChunkOutOfBounds { offset: sb.manifest_offset, length: sb.manifest_length, file_len: store.len(), }); } enforce_limit(sb.manifest_length, MAX_MANIFEST_BYTES)?; read_chunk_bytes(store, sb.manifest_offset, sb.manifest_length) } /// Reads `length` bytes at `offset`, bounds-checking against the store size and /// reporting an out-of-range reference rather than an opaque I/O error. fn read_chunk_bytes( store: &dyn BlockStore, offset: u64, length: u64, ) -> Result, BundleError> { let end = offset.checked_add(length); match end { Some(end) if end <= store.len() => Ok(read_vec(store, offset, length)?), _ => Err(BundleError::ChunkOutOfBounds { offset, length, file_len: store.len(), }), } } /// Writes a chunk payload at `*cursor`, advances the cursor, and returns the /// resulting reference. v0 writes uncompressed, so compressed and uncompressed /// lengths are equal. fn append_chunk( store: &mut dyn BlockStore, cursor: &mut u64, kind: ChunkKind, schema: SchemaVersion, payload: &[u8], ) -> Result { let id = chunk_id(kind, schema, payload); let offset = *cursor; store.write_at(offset, payload)?; *cursor += payload.len() as u64; Ok(ChunkRef { id, kind, schema_version: schema, offset, compressed_length: payload.len() as u64, uncompressed_length: payload.len() as u64, compression: CompressionAlgorithm::None, hash: id.content_hash(), }) } /// Parses one slot for ordinary selection and, if it is a valid committed /// superblock, verifies its manifest chunk's hash against the store. Returns the /// superblock only if every check passes; records a non-committed slot as an /// anomaly. fn verified_slot( store: &dyn BlockStore, slot_bytes: &[u8], anomalies: &mut Vec, ) -> Option { match Superblock::parse_slot(slot_bytes) { SlotParse::Valid(sb) => { // Verify the manifest chunk's hash (Chapter 8 §"Superblock // Selection", step 2): a slot whose manifest is out of the body, // oversize, or does not hash-verify is not valid for ordinary // selection. match read_manifest_payload(store, &sb) { Ok(payload) => { let actual = chunk_content_hash( ChunkKind::Manifest, sb.manifest_schema_version, &payload, ); if actual == sb.manifest_hash { Some(sb) } else { None } } Err(_) => None, } } SlotParse::Rejected(SlotReject::NotCommitted) => { anomalies.push(IntegrityAnomaly::NonCommittedSlot); None } SlotParse::Rejected(_) => None, } } // Re-exported helper for harnesses that build raw images: the body start offset // and the content hash of a manifest payload. /// The content hash of a manifest chunk payload at the **baseline** /// [`Manifest::SCHEMA`] version (Chapter 8 §"Content Hashing"). See /// [`manifest_chunk_hash_versioned`] for a manifest stamped at a /// non-baseline aggregate (G-minor, `spec/PLAN_GMINOR_SCHEMA_MINOR.md` §4, /// pin 7). pub fn manifest_chunk_hash(payload: &[u8]) -> ContentHash { chunk_content_hash(ChunkKind::Manifest, Manifest::SCHEMA, payload) } /// As [`manifest_chunk_hash`], but against the given `schema` rather than the /// baseline [`Manifest::SCHEMA`] — the hash a manifest naming a post-baseline /// edit-barrier tag must be content-addressed under. pub fn manifest_chunk_hash_versioned(payload: &[u8], schema: SchemaVersion) -> ContentHash { chunk_content_hash(ChunkKind::Manifest, schema, payload) } #[cfg(test)] mod tests { use super::*; use crate::ids::{DocumentId, FrontierBytes, ReductionAlgorithmVersion, SnapshotId}; use crate::manifest::SnapshotRef; #[test] fn schema_major_1_admission_is_raised_per_role_op_blocks_only() { // Admission of major 1 is raised **per chunk role**, never as a blanket // accept-set (Binary Format companion §"Schema Major 1"). D2 raised the // operation-envelope-block role to major 1 (a block bearing a v1 // CreateRegion), and schema major 2 to major 2 (a block bearing a v2 // cross-cutting/staff/metadata value); every other role stays exact-0 // until its own versioned path lands, and the manifest stays major 0 // forever. Genesis tranche G2b (`spec/CONTRACT_GENESIS_G2B_TUNING.md`) // raises the op-block role to major 3: `SetTuningContext` is the sole // operation payload that embeds the tuning context, born at major 3 // unconditionally, so a block carrying one is now born at v3 — this // supersedes the earlier Push 4b tranche 3b-i note claiming no // payload ever would. assert_eq!(SchemaVersion::V1.major, 1); assert_eq!(SchemaVersion::V2.major, 2); assert_eq!(SchemaVersion::V3.major, 3); // (t4) The op-block role admits [0, 3] as of genesis tranche G2b. assert_eq!(max_supported_major(ChunkKind::OperationEnvelopeBlock), 3); // The snapshot role admits the major-3 acceleration form (decoded // through the core versioned seam); the canonical BASE stays major 0 // per role (`mis_stamped_canonical_base`). The remaining roles stay // at the generic baseline: the layout cache and the operation index. assert_eq!(SUPPORTED_SCHEMA_MAJOR, 0); assert_eq!(max_supported_major(ChunkKind::Snapshot), 3); assert_eq!(max_supported_major(ChunkKind::LayoutCache), 0); assert_eq!(max_supported_major(ChunkKind::OperationIndex), 0); // The manifest gate is exact to the manifest's own major (0), independent // of the per-role chunk accept-set. assert_eq!(Manifest::SCHEMA.major, 0); } /// (t10) `CONTRACT_GENESIS_G2B_TUNING.md` pin 3: the doc comment above /// `max_supported_major` used to assert a rationale G2b falsifies (no /// operation payload reaches the tuning context, so no op block would /// ever reach schema major 3). That claim must be gone from the source, /// not merely superseded in prose elsewhere; a stale rationale beside a /// corrected constant is exactly how `binary_format.tex:2373` decayed. /// /// **Mutation:** restore the stale sentence into the doc comment; must /// fail. #[test] fn accept_set_doc_no_longer_claims_no_payload_embeds_the_tuning_context() { // BOTH sources, not just this one. The first version of this guard // read only `bundle.rs`, and an identical falsehood survived in // `ids.rs`'s `V3` doc comment precisely because a file-scoped grep can // only ever prove the file it greps. A sibling cross-reference in prose // does not make the pair travel together; sharing one guard does. let stale_claim: String = ["no operation payload ", "embeds the tuning context"].concat(); let stale_consequence: String = ["no op block is ever ", "born at v3"].concat(); for (name, source) in [ ("bundle.rs", include_str!("bundle.rs")), ("ids.rs", include_str!("ids.rs")), ] { assert!( !source.contains(&stale_claim), "{name} still asserts the falsified claim about which payloads carry the tuning context" ); assert!( !source.contains(&stale_consequence), "{name} still asserts the falsified consequence about op blocks and major 3" ); } } #[test] fn committing_an_unsupported_major_op_root_makes_the_live_bundle_read_only() { // A commit publishes an op block beyond this reader's accept-set (a // forward-compat write): structural validation lets it through, but the // LIVE bundle must go read-only at once — not only on the next reopen — // so no further commit runs against canonical history it cannot parse. // // Major 4, not 3: genesis tranche G2b raised the op-block accept-set // to [0, 3] (`SetTuningContext` is born at major 3), so major 3 is now // admitted and this test's "future major" must move past it to stay // an actual test of the read-only-on-overflow path. let mut bundle = fresh_bundle(); let block = StagedChunk::operation_block_versioned( crate::block::encode_block(&[vec![1u8, 2, 3]]), SchemaVersion::new(4, 0), ); bundle .commit(&[block], |ctx| { let mut m = ctx.previous_manifest.clone(); m.operation_roots.push(ctx.new_chunks[0]); m }) .expect("a structurally-valid future-major root is publishable"); assert!( bundle.is_read_only(), "the live bundle is read-only immediately after the commit" ); assert!(bundle.anomalies().iter().any(|a| matches!( a, IntegrityAnomaly::UnsupportedCanonicalChunkMajor { schema_major: 4 } ))); // A further commit against the now-read-only bundle is refused. let more = StagedChunk::operation_block_versioned( crate::block::encode_block(&[vec![9u8]]), SchemaVersion::V0, ); assert!(matches!( bundle.commit(&[more], |ctx| ctx.previous_manifest.clone()), Err(BundleError::ReadOnly) )); } #[test] fn a_canonical_base_stamped_above_major_0_opens_read_only() { // The canonical base is role-bound to major 0 (Binary Format §Schema // Major 2: the MaterializedState embeds no data-model-major values, // and re-stamping byte-identical content churns its address). With // the Snapshot KIND now admitting the major-2 acceleration form, the // per-role check (`mis_stamped_canonical_base`) must catch a // mis-stamped base. // // CONTRACT_FORMAT_EPOCH_MAJOR1.md pin 3a (the "named trap"): this test // used to commit the mis-stamped base directly and assert read-only // immediately, both at commit and at reopen. Neither half is // reachable any more — pin 3a's epoch guard (rows 2/3/5i/6i, both // exercised above in this module) now refuses *any* canonical base, // mis-stamped or not, before `mis_stamped_canonical_base` is ever // consulted, at both `open` and `commit`. That guard is a different // axis (container format major) from this one (data-model schema // major) and does not contradict pin 4 — `ReductionAuthorityUnavailable` // must not degrade to read-only — so the fix is not to weaken pin 4 // toward this test. It is to test what remains reachable: // `mis_stamped_canonical_base` itself, a pure function of a // `Manifest`, independent of the now-categorical bundle-lifecycle // refusal that sits in front of it. let mut m = Manifest::empty(DocumentId([9; 16])); let payload = vec![7u8, 7, 7]; let hash = chunk_content_hash(ChunkKind::Snapshot, SchemaVersion::V1, &payload); let root = ChunkRef { id: ChunkId(hash), kind: ChunkKind::Snapshot, schema_version: SchemaVersion::V1, offset: BODY_START, compressed_length: payload.len() as u64, uncompressed_length: payload.len() as u64, compression: CompressionAlgorithm::None, hash, }; m.canonical_base = Some(SnapshotRef { snapshot_id: SnapshotId([1; 16]), covers_causal_frontier: FrontierBytes::empty(), reduction_algorithm_version: ReductionAlgorithmVersion(0), profile_id: ProfileId::Full, hash: root.hash, root, }); assert_eq!( mis_stamped_canonical_base(&m), Some(1), "a canonical base root stamped above major 0 is still flagged by the per-role \ check, even though pin 3a's epoch guard now intercepts every base before this \ check ever runs" ); } fn fresh_bundle() -> Bundle { Bundle::create( MemStore::new(), FileUuid([1; 16]), Manifest::empty(DocumentId([2; 16])), ) .unwrap() } // ----------------------------------------------------------------- // CONTRACT_FORMAT_EPOCH_MAJOR1.md: format-epoch matrix (pins 1-4). // // After this rung no `create`/`commit` path may ever produce a // base-bearing bundle (pin 3c): `create` already refused one, and rows // 3/6i below now refuse committing one into either epoch, so a // base-bearing fixture can only be built as a hand-crafted image, the // same technique `craft_image`/`craft_image_with_manifest_bytes` already // use further down this file. // ----------------------------------------------------------------- /// A fixed header at `format_major`, bypassing `FixedHeader::new` (which /// always stamps the *current* epoch) so a legacy (major-0) fixture can /// be built too. Same technique as `craft_image` below: public fields, /// public `encode()`. fn header_for_major(format_major: u16, file_uuid: FileUuid) -> FixedHeader { let epoch = if format_major == 0 { FormatEpoch::Legacy } else { FormatEpoch::Current }; FixedHeader { format_major, format_minor: 0, header_length: crate::header::HEADER_LEN as u32, superblock_a_offset: SLOT_A_OFFSET, superblock_b_offset: SLOT_B_OFFSET, file_uuid, epoch, } } /// A minimal valid bundle image at `format_major`, with no canonical base /// (matrix rows 1 and 4). fn craft_image_epoch(format_major: u16) -> Vec { let mut m = Manifest::empty(DocumentId([1; 16])); m.profile_declarations = vec![ProfileDeclaration::full()]; let payload = m.encode(); let mut image = vec![0u8; BODY_START as usize]; image.extend_from_slice(&payload); let sb = Superblock { generation: 0, manifest_offset: BODY_START, manifest_length: payload.len() as u64, manifest_hash: manifest_chunk_hash(&payload), manifest_schema_version: SchemaVersion::V0, reduction_algorithm_version: ReductionAlgorithmVersion(0), profile_id: ProfileId::Full, commit_state: CommitState::Committed, commit_timestamp: WallClockTime(0), }; image[0..crate::header::HEADER_LEN as usize] .copy_from_slice(&header_for_major(format_major, FileUuid([1; 16])).encode()); image[SLOT_A_OFFSET as usize..SLOT_A_OFFSET as usize + SUPERBLOCK_LEN as usize] .copy_from_slice(&sb.encode()); image } /// A bundle image at `format_major` carrying a canonical base wired /// directly into the manifest and superblock bytes, bypassing /// `create`/`commit` entirely — the only way left to build this fixture /// (pin 3c). `base_schema_major` mis-stamps the base's own root /// independent of the container's format major; `base_version` and /// `superblock_version` are equal for a self-consistent base (rows /// 2/5i/11) and unequal for the corrupt-base precedence fixture (test 7). fn craft_image_with_base( format_major: u16, base_schema_major: u16, base_version: ReductionAlgorithmVersion, superblock_version: ReductionAlgorithmVersion, ) -> Vec { let base_schema = SchemaVersion::new(base_schema_major, 0); let snap_payload = vec![7u8, 7, 7]; let snap_hash = chunk_content_hash(ChunkKind::Snapshot, base_schema, &snap_payload); let root = ChunkRef { id: ChunkId(snap_hash), kind: ChunkKind::Snapshot, schema_version: base_schema, offset: BODY_START, compressed_length: snap_payload.len() as u64, uncompressed_length: snap_payload.len() as u64, compression: CompressionAlgorithm::None, hash: snap_hash, }; let mut m = Manifest::empty(DocumentId([1; 16])); m.profile_declarations = vec![ProfileDeclaration::full()]; m.canonical_base = Some(SnapshotRef { snapshot_id: SnapshotId([1; 16]), covers_causal_frontier: FrontierBytes::empty(), reduction_algorithm_version: base_version, profile_id: ProfileId::Full, hash: root.hash, root, }); let manifest_payload = m.encode(); let manifest_offset = BODY_START + snap_payload.len() as u64; let mut image = vec![0u8; BODY_START as usize]; image.extend_from_slice(&snap_payload); image.extend_from_slice(&manifest_payload); let sb = Superblock { generation: 0, manifest_offset, manifest_length: manifest_payload.len() as u64, manifest_hash: manifest_chunk_hash(&manifest_payload), manifest_schema_version: SchemaVersion::V0, reduction_algorithm_version: superblock_version, profile_id: ProfileId::Full, commit_state: CommitState::Committed, commit_timestamp: WallClockTime(0), }; image[0..crate::header::HEADER_LEN as usize] .copy_from_slice(&header_for_major(format_major, FileUuid([1; 16])).encode()); image[SLOT_A_OFFSET as usize..SLOT_A_OFFSET as usize + SUPERBLOCK_LEN as usize] .copy_from_slice(&sb.encode()); image } /// `Bundle` has no `Debug` impl, so `Result::expect_err`/`unwrap_err` /// cannot be used directly on `Bundle::open`'s result; this extracts the /// error by hand. fn open_err(image: Vec, context: &str) -> BundleError { match Bundle::open(MemStore::from_bytes(image)) { Err(e) => e, Ok(_) => panic!("{context}"), } } #[test] fn a_legacy_major_0_bundle_without_a_base_opens() { // Matrix row 1: a legacy container with no base opens cleanly — // nothing unverifiable is exposed. let image = craft_image_epoch(0); let bundle = Bundle::open(MemStore::from_bytes(image)) .expect("a legacy bundle without a base opens"); assert!(!bundle.is_read_only()); assert!(bundle.anomalies().is_empty()); assert_eq!(bundle.header().epoch, FormatEpoch::Legacy); } #[test] fn a_legacy_major_0_bundle_with_a_base_is_rejected() { // Matrix row 2: a legacy container already carrying a // (self-consistent) base is a hard reject at open — a pre-epoch base // was never validated against a reduction authority. let image = craft_image_with_base( 0, 0, ReductionAlgorithmVersion(0), ReductionAlgorithmVersion(0), ); let err = open_err(image, "a legacy base-bearing bundle must not open"); assert!(matches!(err, BundleError::LegacyBundleHasCanonicalBase)); assert!(!matches!(err, BundleError::LegacyBaseIntroductionRejected)); assert!(!matches!(err, BundleError::ReductionAuthorityUnavailable)); } #[test] fn adding_a_base_to_a_legacy_bundle_is_rejected_and_names_repack() { // Matrix row 3: a legacy container that opened clean under row 1 // must not become base-bearing in place — the epoch is not // inheritable, because its header can never say the base was // validated. let image = craft_image_epoch(0); let mut bundle = Bundle::open(MemStore::from_bytes(image)) .expect("a legacy bundle without a base opens"); let staged = StagedChunk { kind: ChunkKind::Snapshot, schema_version: SchemaVersion::V0, payload: vec![7u8, 7, 7], }; let err = bundle .commit(&[staged], |ctx| { let mut m = ctx.previous_manifest.clone(); let root = ctx.new_chunks[0]; m.canonical_base = Some(SnapshotRef { snapshot_id: SnapshotId([1; 16]), covers_causal_frontier: FrontierBytes::empty(), reduction_algorithm_version: ReductionAlgorithmVersion(0), profile_id: ProfileId::Full, hash: root.hash, root, }); m }) .expect_err("committing a base into a legacy bundle must be refused"); assert!(matches!(err, BundleError::LegacyBaseIntroductionRejected)); assert!(!matches!(err, BundleError::LegacyBundleHasCanonicalBase)); assert!(!matches!(err, BundleError::ReductionAuthorityUnavailable)); assert!( err.to_string().to_lowercase().contains("repack"), "row-3 error must name repack: {err}" ); assert_eq!( bundle.generation(), 0, "the refused commit did not advance the bundle" ); } #[test] fn a_major_1_bundle_round_trips_and_refuses_to_introduce_a_base() { // Matrix row 4 (round-trip half) + row 6i (refusal half). Renamed // from "..._validates_its_base", which pin 3a forbids claiming until // P13-S27 lands: this rung does not validate a base, it refuses one // outright. let mut bundle = fresh_bundle(); assert_eq!(bundle.header().epoch, FormatEpoch::Current); // Round trip: ordinary non-base-bearing history commits and reopens. let block = StagedChunk::operation_block(block::encode_block(&[vec![1u8, 2, 3]])); bundle .commit(&[block], |ctx| { let mut m = ctx.previous_manifest.clone(); m.operation_roots.push(ctx.new_chunks[0]); m }) .expect("a major-1 bundle commits ordinary non-base-bearing history"); let image = bundle.into_store().into_bytes(); let mut reopened = Bundle::open(MemStore::from_bytes(image)).expect("a major-1 bundle round trips"); assert!(!reopened.is_read_only()); assert_eq!(reopened.manifest().operation_roots.len(), 1); // Refusal: introducing a base is refused (row 6i), distinctly from // both legacy errors. let staged = StagedChunk { kind: ChunkKind::Snapshot, schema_version: SchemaVersion::V0, payload: vec![9u8, 9, 9], }; let err = reopened .commit(&[staged], |ctx| { let mut m = ctx.previous_manifest.clone(); let root = ctx.new_chunks[0]; m.canonical_base = Some(SnapshotRef { snapshot_id: SnapshotId([2; 16]), covers_causal_frontier: FrontierBytes::empty(), reduction_algorithm_version: ReductionAlgorithmVersion(0), profile_id: ProfileId::Full, hash: root.hash, root, }); m }) .expect_err( "a major-1 bundle must refuse to introduce a base during the S28 -> P13-S27 interval", ); assert!(matches!(err, BundleError::ReductionAuthorityUnavailable)); assert!(!matches!(err, BundleError::LegacyBundleHasCanonicalBase)); assert!(!matches!(err, BundleError::LegacyBaseIntroductionRejected)); } #[test] fn a_corrupt_base_fails_as_malformed_before_any_epoch_error() { // Pin 3's precedence rule, in BOTH epochs: a base whose version // disagrees with its superblock's is corrupt and must fail with the // existing malformed-bundle error — never row 2's legacy error, and // never row 5i's `ReductionAuthorityUnavailable`. Collapsing // tampering into staleness would erase the distinction P13-S27's // authority check rests on. Renamed from `..._corrupt_legacy_base...`: // the major-1 half is the one an earlier draft missed. for format_major in [0u16, 1u16] { let image = craft_image_with_base( format_major, 0, ReductionAlgorithmVersion(1), ReductionAlgorithmVersion(2), // disagrees with the base's own version ); let err = open_err(image, "a corrupt base must not open"); assert!( matches!(err, BundleError::Decode(DecodeError::Malformed(_))), "format_major {format_major}: expected the malformed-bundle error, got {err:?}" ); assert!(!matches!(err, BundleError::LegacyBundleHasCanonicalBase)); assert!(!matches!(err, BundleError::ReductionAuthorityUnavailable)); } } #[test] fn opening_a_major_1_bundle_that_already_carries_a_base_is_refused() { // Matrix row 5i, the read-side branch pin 3a adds — an earlier draft // omitted it entirely. let image = craft_image_with_base( 1, 0, ReductionAlgorithmVersion(0), ReductionAlgorithmVersion(0), ); let err = open_err( image, "a major-1 bundle already carrying a base must not open during the interval", ); assert!(matches!(err, BundleError::ReductionAuthorityUnavailable)); assert!(!matches!(err, BundleError::LegacyBundleHasCanonicalBase)); assert!(!matches!(err, BundleError::LegacyBaseIntroductionRejected)); } // ----------------------------------------------------------------- // G-minor (`spec/PLAN_GMINOR_SCHEMA_MINOR.md` §4): s10, s11, s12. // ----------------------------------------------------------------- #[test] fn s10_a_repack_with_unchanged_barrier_content_preserves_the_carried_version() { // A "repack" here: a second commit whose closure carries forward the // manifest completely unchanged (no barrier content changed), reading // the exact prior aggregate from `CommitContext::previous_manifest_ // version` (pin 6.2/6.6) rather than recomputing it. The carried // version must survive byte-for-byte. let mut bundle = Bundle::create_versioned( MemStore::new(), FileUuid([9; 16]), Manifest::empty(DocumentId([9; 16])), SchemaVersion::new(0, 8), ) .unwrap(); assert_eq!( bundle.superblock().manifest_schema_version, SchemaVersion::new(0, 8) ); let mut observed_previous = None; bundle .commit_versioned(&[], SchemaVersion::new(0, 8), |ctx| { observed_previous = Some(ctx.previous_manifest_version); ctx.previous_manifest.clone() }) .unwrap(); assert_eq!( observed_previous, Some(SchemaVersion::new(0, 8)), "CommitContext must expose the previous manifest version for the closure to read" ); assert_eq!( bundle.superblock().manifest_schema_version, SchemaVersion::new(0, 8), "an unchanged-content repack preserves the carried version exactly" ); } #[test] fn s11_bundle_rs_301_accepts_a_manifest_minor_mismatch_at_the_same_major() { // `open`'s gate at :301 compares `.major` only. A bundle whose // manifest minor differs from `Manifest::SCHEMA`'s (but whose major // matches) must still open — the v0 rule that the minor is a record, // not a gate (pin 7). This is the exact conformance regression pin 7 // warns tightening the comparison to full-version equality would // cause. let bundle = Bundle::create_versioned( MemStore::new(), FileUuid([10; 16]), Manifest::empty(DocumentId([10; 16])), SchemaVersion::new(0, 8), // differs from Manifest::SCHEMA's minor (1) ) .unwrap(); assert_ne!(SchemaVersion::new(0, 8), Manifest::SCHEMA); assert_eq!(SchemaVersion::new(0, 8).major, Manifest::SCHEMA.major); let image = bundle.into_store().into_bytes(); let reopened = Bundle::open(MemStore::from_bytes(image)) .expect("a major-matching minor mismatch opens"); assert!(!reopened.is_read_only()); assert_eq!( reopened.superblock().manifest_schema_version, SchemaVersion::new(0, 8) ); } #[test] fn s12_a_raised_minor_changes_the_chunk_id_and_therefore_the_manifest_id() { // Two operation blocks with byte-identical payloads but different // schema minors must have different `ChunkId`s — the minor enters the // hash preimage (`canonical_bytes`: major then minor). Naming that // `ChunkRef` from a manifest's `operation_roots` then changes the // manifest body, and therefore the `ManifestId`. let payload = vec![1u8, 2, 3]; let id_v0 = crate::chunk::chunk_id( ChunkKind::OperationEnvelopeBlock, SchemaVersion::V0, &payload, ); let id_minor_8 = crate::chunk::chunk_id( ChunkKind::OperationEnvelopeBlock, SchemaVersion::new(0, 8), &payload, ); assert_ne!( id_v0, id_minor_8, "raising the minor must change the ChunkId" ); let make_manifest = |chunk_id: epiphany_determinism::ChunkId, schema: SchemaVersion| { let mut m = Manifest::empty(DocumentId([11; 16])); m.operation_roots = vec![ChunkRef { id: chunk_id, kind: ChunkKind::OperationEnvelopeBlock, schema_version: schema, offset: 0, compressed_length: payload.len() as u64, uncompressed_length: payload.len() as u64, compression: CompressionAlgorithm::None, hash: chunk_id.content_hash(), }]; m }; let manifest_v0 = make_manifest(id_v0, SchemaVersion::V0); let manifest_minor_8 = make_manifest(id_minor_8, SchemaVersion::new(0, 8)); assert_ne!( manifest_v0.derive_id(), manifest_minor_8.derive_id(), "a changed ChunkRef in operation_roots must change the ManifestId" ); } #[test] fn create_then_open_round_trips() { let bundle = fresh_bundle(); let image = bundle.into_store().into_bytes(); let reopened = Bundle::open(MemStore::from_bytes(image)).unwrap(); assert_eq!(reopened.generation(), 0); assert_eq!(reopened.active_slot(), Slot::A); assert_eq!(reopened.file_uuid(), FileUuid([1; 16])); assert!(!reopened.is_read_only()); assert!(reopened.anomalies().is_empty()); } #[test] fn commit_advances_generation_and_flips_slot() { let mut bundle = fresh_bundle(); let block = block::encode_block(&[b"env-1".to_vec(), b"env-2".to_vec()]); bundle .commit(&[StagedChunk::operation_block(block)], |ctx| { let mut m = ctx.previous_manifest.clone(); m.operation_roots = ctx.new_chunks.to_vec(); m }) .unwrap(); assert_eq!(bundle.generation(), 1); assert_eq!(bundle.active_slot(), Slot::B); // Reopen from the image; the committed state is visible and verifies. let image = bundle.into_store().into_bytes(); let reopened = Bundle::open(MemStore::from_bytes(image)).unwrap(); assert_eq!(reopened.generation(), 1); assert_eq!(reopened.manifest().operation_roots.len(), 1); reopened.verify_canonical_chunks().unwrap(); let envs = reopened .read_operation_block(&reopened.manifest().operation_roots[0]) .unwrap(); assert_eq!(envs, vec![b"env-1".to_vec(), b"env-2".to_vec()]); } #[test] fn successive_commits_alternate_slots() { let mut bundle = fresh_bundle(); for i in 1..=5u64 { let payload = block::encode_block(&[vec![i as u8; 16]]); bundle .commit(&[StagedChunk::operation_block(payload)], |ctx| { let mut m = ctx.previous_manifest.clone(); m.operation_roots.extend(ctx.new_chunks.iter().copied()); m }) .unwrap(); assert_eq!(bundle.generation(), i); assert_eq!( bundle.active_slot(), if i % 2 == 1 { Slot::B } else { Slot::A } ); } assert_eq!(bundle.manifest().operation_roots.len(), 5); } #[test] fn a_corrupt_canonical_chunk_is_hard_corruption() { let mut bundle = fresh_bundle(); let payload = block::encode_block(&[b"env".to_vec()]); bundle .commit(&[StagedChunk::operation_block(payload)], |ctx| { let mut m = ctx.previous_manifest.clone(); m.operation_roots = ctx.new_chunks.to_vec(); m }) .unwrap(); let op_offset = bundle.manifest().operation_roots[0].offset; let mut image = bundle.into_store().into_bytes(); image[op_offset as usize] ^= 0xFF; // corrupt the operation block payload let reopened = Bundle::open(MemStore::from_bytes(image)).unwrap(); // Opening still works (the manifest/superblock are intact)... assert_eq!(reopened.generation(), 1); // ...but verifying the canonical chunk surfaces hard corruption. assert!(matches!( reopened.verify_canonical_chunks(), Err(BundleError::ChunkHashMismatch { .. }) )); } #[test] fn commit_is_refused_on_a_read_only_bundle() { let mut bundle = fresh_bundle(); bundle.read_only = true; let err = bundle.commit(&[], |ctx| ctx.previous_manifest.clone()); assert!(matches!(err, Err(BundleError::ReadOnly))); } fn op_block(bundle: &mut Bundle, envelopes: &[&[u8]]) { let payload = block::encode_block(&envelopes.iter().map(|e| e.to_vec()).collect::>()); bundle .commit(&[StagedChunk::operation_block(payload)], |ctx| { let mut m = ctx.previous_manifest.clone(); m.operation_roots.extend(ctx.new_chunks.iter().copied()); m }) .unwrap(); } #[test] fn commit_rejects_a_dangling_canonical_root() { // Finding 2: a builder closure that publishes a root pointing at a chunk // that does not exist must be refused; the bundle stays at generation 0. let mut bundle = fresh_bundle(); let bogus = ChunkRef { id: ChunkId(ContentHash([9; 32])), kind: ChunkKind::OperationEnvelopeBlock, schema_version: SchemaVersion::V0, offset: 1_000_000, compressed_length: 10, uncompressed_length: 10, compression: CompressionAlgorithm::None, hash: ContentHash([9; 32]), }; let err = bundle.commit(&[], |ctx| { let mut m = ctx.previous_manifest.clone(); m.operation_roots = vec![bogus]; m }); assert!(matches!(err, Err(BundleError::ChunkOutOfBounds { .. }))); assert_eq!(bundle.generation(), 0); } #[test] fn commit_rejects_a_root_with_a_mismatched_hash() { // Finding 2: a root referencing a real chunk but with the wrong declared // hash is rejected before commit. let mut bundle = fresh_bundle(); op_block(&mut bundle, &[b"real"]); let mut tampered = bundle.manifest().operation_roots[0]; tampered.hash = ContentHash([0; 32]); let err = bundle.commit(&[], |ctx| { let mut m = ctx.previous_manifest.clone(); m.operation_roots = vec![tampered]; m }); assert!(matches!(err, Err(BundleError::ChunkHashMismatch { .. }))); } #[test] fn create_rejects_initial_canonical_roots() { // Finding 2: create() writes no chunks, so any canonical root would be // dangling; reject it. let mut m = Manifest::empty(DocumentId([5; 16])); m.operation_roots = vec![ChunkRef { id: ChunkId(ContentHash([1; 32])), kind: ChunkKind::OperationEnvelopeBlock, schema_version: SchemaVersion::V0, offset: BODY_START, compressed_length: 1, uncompressed_length: 1, compression: CompressionAlgorithm::None, hash: ContentHash([1; 32]), }]; assert!(Bundle::create(MemStore::new(), FileUuid([1; 16]), m).is_err()); } #[test] fn forged_chunk_id_is_rejected_on_read() { // Finding 12: a reference whose id disagrees with its (correct) hash // fails verification. let mut bundle = fresh_bundle(); op_block(&mut bundle, &[b"env"]); let mut forged = bundle.manifest().operation_roots[0]; forged.id = ChunkId(ContentHash([0xAB; 32])); // hash still correct assert!(matches!( bundle.read_chunk(&forged), Err(BundleError::ChunkHashMismatch { .. }) )); } #[test] fn manifest_id_is_synced_after_create_and_commit() { // Finding 13: the in-memory manifest id matches the derived id, with no // reopen required. let mut bundle = fresh_bundle(); assert_eq!(bundle.manifest().manifest_id, bundle.manifest().derive_id()); op_block(&mut bundle, &[b"env"]); assert_eq!(bundle.manifest().manifest_id, bundle.manifest().derive_id()); } fn commit_blob(bundle: &mut Bundle, payload: &[u8]) { let staged = StagedChunk { kind: ChunkKind::Blob, schema_version: SchemaVersion::V0, payload: payload.to_vec(), }; bundle .commit(&[staged], |ctx| { let mut m = ctx.previous_manifest.clone(); let r = ctx.new_chunks[0]; m.blob_roots.push(BlobRef { blob_id: BlobId(r.hash), media_type: "application/octet-stream".to_string(), offset: r.offset, compressed_length: r.compressed_length, uncompressed_length: r.uncompressed_length, compression: CompressionAlgorithm::None, hash: r.hash, declared_max_uncompressed_length: None, }); m }) .unwrap(); } #[test] fn staged_blob_reads_back() { // Finding 5: a staged Blob chunk hashes the way it is verified, so it // round-trips through read_blob (wired into blob_roots, not the // operation roots — which would now be rejected for the wrong kind). let mut bundle = fresh_bundle(); let payload = b"blob-bytes".to_vec(); commit_blob(&mut bundle, &payload); let b = bundle.manifest().blob_roots[0].clone(); assert_eq!(bundle.read_blob(&b).unwrap(), payload); } #[test] fn duplicate_content_is_deduplicated() { // Committing the same payload twice reuses the existing chunk's storage // (storage dedup) and collapses to a single canonical root (manifest // dedup), with the in-memory manifest already normalized. let mut bundle = fresh_bundle(); op_block(&mut bundle, &[b"same-content"]); let cursor_before = bundle.write_cursor; let manifest_len_before = bundle.superblock().manifest_length; op_block(&mut bundle, &[b"same-content"]); // build extends roots to [r, r] // Manifest dedup: the duplicate operation root collapsed to one, in // memory (not only after a reopen). assert_eq!(bundle.manifest().operation_roots.len(), 1); // Storage dedup: the only body growth was the new manifest chunk, not a // second copy of the block. let grew = bundle.write_cursor - cursor_before; assert!( grew <= manifest_len_before + 256, "duplicate content appears to have been re-stored (body grew by {grew})" ); // And a reopen agrees with the in-memory (already normalized) manifest. let image = bundle.into_store().into_bytes(); let reopened = Bundle::open(MemStore::from_bytes(image)).unwrap(); assert_eq!(reopened.manifest().operation_roots.len(), 1); } #[test] fn required_extension_forces_read_only() { // Finding 3: a declared required extension (unknown in v0) opens the // bundle read-only. let mut bundle = fresh_bundle(); bundle .commit(&[], |ctx| { let mut m = ctx.previous_manifest.clone(); m.extension_declarations .push(crate::manifest::ExtensionDeclaration { extension_id: crate::ids::ExtensionId([3; 16]), version: crate::ids::SemVer::new(1, 0, 0), required: true, preserved_chunk_roots: Vec::new(), affected_object_kinds: Vec::new(), edit_barriers: Vec::new(), }); m }) .unwrap(); let image = bundle.into_store().into_bytes(); let reopened = Bundle::open(MemStore::from_bytes(image)).unwrap(); assert!(reopened.is_read_only()); assert!(reopened .anomalies() .contains(&IntegrityAnomaly::UnknownRequiredExtension)); } #[test] fn commit_preserves_unknown_extension_roots_when_the_closure_drops_them() { // The bundle's job is preservation: an extension-*unaware* writer that // rebuilds the manifest from scratch must not orphan an unknown // (optional) extension's preserved roots. let mut bundle = fresh_bundle(); let ext_id = crate::ids::ExtensionId([7; 16]); let ext_root = ChunkRef { id: ChunkId(ContentHash([42; 32])), kind: ChunkKind::ExtensionData, schema_version: SchemaVersion::V0, offset: 4096, compressed_length: 8, uncompressed_length: 8, compression: CompressionAlgorithm::None, hash: ContentHash([42; 32]), }; // 1) An extension-aware commit declares the optional extension + a root. bundle .commit(&[], |ctx| { let mut m = ctx.previous_manifest.clone(); m.extension_declarations .push(crate::manifest::ExtensionDeclaration { extension_id: ext_id, version: crate::ids::SemVer::new(1, 0, 0), required: false, preserved_chunk_roots: vec![ext_root], affected_object_kinds: Vec::new(), edit_barriers: Vec::new(), }); m }) .unwrap(); assert!(bundle .manifest() .extension_declarations .iter() .any(|e| e.extension_id == ext_id)); // 2) An extension-unaware commit rebuilds the manifest from empty, // carrying only what it understands (operation roots). The bundle must // still carry the extension and its root forward. let doc = bundle.manifest().document_id; bundle .commit(&[], |ctx| { let mut m = Manifest::empty(doc); m.operation_roots = ctx.previous_manifest.operation_roots.clone(); m }) .unwrap(); let survives = |m: &Manifest| { m.extension_declarations.iter().any(|e| { e.extension_id == ext_id && e.preserved_chunk_roots.iter().any(|r| r.id == ext_root.id) }) }; assert!( survives(bundle.manifest()), "unknown extension + its root must survive an extension-unaware commit" ); // 3) And it survives a reopen (durably preserved). let image = bundle.into_store().into_bytes(); let reopened = Bundle::open(MemStore::from_bytes(image)).unwrap(); assert!(survives(reopened.manifest())); } #[test] fn profile_selection_is_stable_across_reload() { // Finding 9: the superblock's profile_id is the canonical-first profile, // so a [Lite, Full] manifest does not flip its active profile on reload. use crate::manifest::{ProfileConstraints, ProfileDeclaration}; let mut m = Manifest::empty(DocumentId([6; 16])); m.profile_declarations = vec![ ProfileDeclaration { profile_id: ProfileId::Lite, version: crate::ids::SemVer::new(0, 1, 0), constraints: ProfileConstraints::DEFAULT_FULL, }, ProfileDeclaration { profile_id: ProfileId::Full, version: crate::ids::SemVer::new(0, 1, 0), constraints: ProfileConstraints::DEFAULT_FULL, }, ]; let bundle = Bundle::create(MemStore::new(), FileUuid([1; 16]), m).unwrap(); let in_memory_profile = bundle.superblock().profile_id; let image = bundle.into_store().into_bytes(); let reopened = Bundle::open(MemStore::from_bytes(image)).unwrap(); assert_eq!(in_memory_profile, reopened.superblock().profile_id); } #[test] fn corrupt_canonical_blob_is_surfaced() { // Finding 4: a corrupt blob referenced by the manifest is caught by // verify_canonical_chunks. let mut bundle = fresh_bundle(); commit_blob(&mut bundle, b"audio-bytes"); let blob_offset = bundle.manifest().blob_roots[0].offset; bundle.verify_canonical_chunks().unwrap(); // intact let mut image = bundle.into_store().into_bytes(); image[blob_offset as usize] ^= 0xFF; let reopened = Bundle::open(MemStore::from_bytes(image)).unwrap(); assert!(matches!( reopened.verify_canonical_chunks(), Err(BundleError::ChunkHashMismatch { .. }) )); } #[test] fn empty_profile_list_is_rejected_at_create_and_commit() { // Finding 1: a writer must not emit a manifest its own open would reject. let mut m = Manifest::empty(DocumentId([8; 16])); m.profile_declarations.clear(); assert!(Bundle::create(MemStore::new(), FileUuid([1; 16]), m).is_err()); let mut bundle = fresh_bundle(); let err = bundle.commit(&[], |ctx| { let mut m = ctx.previous_manifest.clone(); m.profile_declarations.clear(); m }); assert!(err.is_err()); assert_eq!(bundle.generation(), 0); } #[test] fn commit_rejects_a_dangling_blob_root() { // Finding 3: a blob root pointing nowhere is refused before commit. let mut bundle = fresh_bundle(); let err = bundle.commit(&[], |ctx| { let mut m = ctx.previous_manifest.clone(); m.blob_roots.push(BlobRef { blob_id: BlobId(ContentHash([4; 32])), media_type: "x/y".to_string(), offset: 9_000_000, compressed_length: 4, uncompressed_length: 4, compression: CompressionAlgorithm::None, hash: ContentHash([4; 32]), declared_max_uncompressed_length: None, }); m }); assert!(err.is_err()); assert_eq!(bundle.generation(), 0); } #[test] fn commit_rejects_wrong_kind_operation_root() { // Finding 2: an operation root must be an operation-envelope block. let mut bundle = fresh_bundle(); let staged = StagedChunk { kind: ChunkKind::LayoutCache, schema_version: SchemaVersion::V0, payload: b"not a block".to_vec(), }; let err = bundle.commit(&[staged], |ctx| { let mut m = ctx.previous_manifest.clone(); m.operation_roots = ctx.new_chunks.to_vec(); m }); assert!(matches!(err, Err(BundleError::Decode(_)))); } #[test] fn live_bundle_becomes_read_only_after_committing_a_required_extension() { // Finding 4: read-only takes effect on the live object, not only on // reopen. let mut bundle = fresh_bundle(); bundle .commit(&[], |ctx| { let mut m = ctx.previous_manifest.clone(); m.extension_declarations .push(crate::manifest::ExtensionDeclaration { extension_id: crate::ids::ExtensionId([3; 16]), version: crate::ids::SemVer::new(1, 0, 0), required: true, preserved_chunk_roots: Vec::new(), affected_object_kinds: Vec::new(), edit_barriers: Vec::new(), }); m }) .unwrap(); assert!(bundle.is_read_only()); assert!(matches!( bundle.commit(&[], |ctx| ctx.previous_manifest.clone()), Err(BundleError::ReadOnly) )); } #[test] fn indeterminate_final_flush_poisons_the_bundle() { // Finding 5: if the commit-point flush persists G+1 but returns an error, // the live bundle must not keep accepting commits against stale state. use crate::store::{CrashPoint, FaultStore, Tear}; // Discover the commit's final-flush syscall index via a no-fault run. let base = fresh_bundle().into_store().into_bytes(); let total = { let mut b = Bundle::open(FaultStore::no_fault(base.clone())).unwrap(); b.commit( &[StagedChunk::operation_block(block::encode_block(&[ b"e".to_vec() ]))], |ctx| { let mut m = ctx.previous_manifest.clone(); m.operation_roots.extend(ctx.new_chunks.iter().copied()); m }, ) .unwrap(); b.into_store().syscalls_issued() }; // Crash on the final flush but persist the whole superblock (full tear): // durable is G+1, yet commit returns Err. let crash = CrashPoint { after_syscalls: total - 1, tear: Tear::TornLastWrite { prefix: 256 }, }; let mut bundle = Bundle::open(FaultStore::new(base, crash)).unwrap(); let result = bundle.commit( &[StagedChunk::operation_block(block::encode_block(&[ b"e".to_vec() ]))], |ctx| { let mut m = ctx.previous_manifest.clone(); m.operation_roots.extend(ctx.new_chunks.iter().copied()); m }, ); assert!(result.is_err(), "the final flush errored"); // The bundle is poisoned: further commits are refused (the durable state // is indeterminate and must be reloaded). assert!(bundle.is_read_only()); } #[test] fn create_with_required_extension_is_read_only_immediately() { // Finding 3: a freshly created bundle with a required extension is // read-only without a reopen. let mut m = Manifest::empty(DocumentId([9; 16])); m.extension_declarations .push(crate::manifest::ExtensionDeclaration { extension_id: crate::ids::ExtensionId([1; 16]), version: crate::ids::SemVer::new(1, 0, 0), required: true, preserved_chunk_roots: Vec::new(), affected_object_kinds: Vec::new(), edit_barriers: Vec::new(), }); let bundle = Bundle::create(MemStore::new(), FileUuid([1; 16]), m).unwrap(); assert!(bundle.is_read_only()); } #[test] fn commit_rejects_a_canonical_base_with_an_undeclared_profile() { // Finding 1: a writer must not emit a canonical base whose profile its // own open would reject. let mut bundle = fresh_bundle(); // declares only Full let snap_payload = b"snapshot".to_vec(); let staged = StagedChunk { kind: ChunkKind::Snapshot, schema_version: SchemaVersion::V0, payload: snap_payload, }; let err = bundle.commit(&[staged], |ctx| { let mut m = ctx.previous_manifest.clone(); let root = ctx.new_chunks[0]; m.canonical_base = Some(crate::manifest::SnapshotRef { snapshot_id: crate::ids::SnapshotId([1; 16]), covers_causal_frontier: crate::ids::FrontierBytes::empty(), reduction_algorithm_version: ReductionAlgorithmVersion(0), profile_id: ProfileId::Lite, // NOT declared by the manifest root, hash: root.hash, }); m }); assert!(matches!(err, Err(BundleError::Decode(_)))); assert_eq!(bundle.generation(), 0); } #[test] fn oversize_manifest_is_refused_by_the_writer() { // Finding 2: a manifest exceeding the reader limit is rejected at // create/commit, not only on reopen. let mut m = Manifest::empty(DocumentId([2; 16])); m.extension_declarations .push(crate::manifest::ExtensionDeclaration { extension_id: crate::ids::ExtensionId([1; 16]), version: crate::ids::SemVer::new(1, 0, 0), required: false, preserved_chunk_roots: Vec::new(), affected_object_kinds: vec![0u8; (MAX_MANIFEST_BYTES + 1) as usize], edit_barriers: Vec::new(), }); assert!(matches!( Bundle::create(MemStore::new(), FileUuid([1; 16]), m), Err(BundleError::ResourceLimitExceeded { .. }) )); } #[test] fn block_size_limit_follows_the_active_superblock_profile() { // Finding 4: a bundle selected under a smaller profile reads blocks under // that profile's limit, not the canonical-first profile's. use crate::manifest::{ProfileConstraints, ProfileDeclaration, RetentionPolicy}; let big = ProfileConstraints { max_uncompressed_block_size: 64 << 20, retention_policy: RetentionPolicy::DEFAULT_FULL, }; let small = ProfileConstraints { max_uncompressed_block_size: 2048, retention_policy: RetentionPolicy::DEFAULT_FULL, }; let mut m = Manifest::empty(DocumentId([3; 16])); m.profile_declarations = vec![ ProfileDeclaration { profile_id: ProfileId::Full, version: crate::ids::SemVer::new(0, 1, 0), constraints: big, }, ProfileDeclaration { profile_id: ProfileId::Lite, version: crate::ids::SemVer::new(0, 1, 0), constraints: small, }, ]; let mut bundle = Bundle::create(MemStore::new(), FileUuid([1; 16]), m).unwrap(); // Canonical-first profile is Full (discriminant 0): active = Full limit. assert_eq!(bundle.superblock().profile_id, ProfileId::Full); assert_eq!(bundle.max_block_size(), 64 << 20); // Simulate a bundle selected under Lite (as a foreign writer might emit): // the active limit must follow the superblock, not canonical-first. bundle.superblock.profile_id = ProfileId::Lite; assert_eq!(bundle.max_block_size(), 2048); } fn profile( profile_id: ProfileId, version: crate::ids::SemVer, max_block: u64, ) -> ProfileDeclaration { ProfileDeclaration { profile_id, version, constraints: crate::manifest::ProfileConstraints { max_uncompressed_block_size: max_block, retention_policy: crate::manifest::RetentionPolicy::DEFAULT_FULL, }, } } #[test] fn profile_support_is_classified_correctly() { // Findings 2/3/6: built-in editable, ReadOnly understood-but-not-editable, // Custom/future-major/oversize-block not understood. let v0 = crate::ids::SemVer::new(0, 1, 0); let full = profile(ProfileId::Full, v0, 1 << 20); assert!(profile_is_editable(&full)); let ro = profile(ProfileId::ReadOnly, v0, 1 << 20); assert!(profile_is_understood(&ro) && !profile_is_editable(&ro)); let custom = profile( ProfileId::Custom(crate::ids::ProfileRegistryId([1; 16])), v0, 1 << 20, ); assert!(!profile_is_understood(&custom)); let future = profile(ProfileId::Full, crate::ids::SemVer::new(1, 0, 0), 1 << 20); assert!(!profile_is_understood(&future)); let huge = profile(ProfileId::Full, v0, MAX_CHUNK_BYTES + 1); assert!(!profile_is_understood(&huge)); } #[test] fn create_rejects_unsupported_or_duplicate_active_profiles() { let v0 = crate::ids::SemVer::new(0, 1, 0); let make = |decls: Vec| { let mut m = Manifest::empty(DocumentId([1; 16])); m.profile_declarations = decls; Bundle::create(MemStore::new(), FileUuid([1; 16]), m) }; // Custom-only (no understood profile to operate under). assert!(make(vec![profile( ProfileId::Custom(crate::ids::ProfileRegistryId([2; 16])), v0, 1 << 20 )]) .is_err()); // Block bound beyond the reader's hard limit (unsupported). assert!(make(vec![profile(ProfileId::Full, v0, MAX_CHUNK_BYTES + 1)]).is_err()); // Duplicate profile id. assert!(make(vec![ profile(ProfileId::Full, v0, 1 << 20), profile(ProfileId::Full, crate::ids::SemVer::new(0, 2, 0), 1 << 20), ]) .is_err()); // A plain Full profile is fine. assert!(make(vec![profile(ProfileId::Full, v0, 1 << 20)]).is_ok()); } #[test] fn read_only_profile_is_emittable_as_a_read_only_bundle() { // Finding: a sole ReadOnly profile is a *valid* bundle to produce — it // just opens read-only (the spec describes ReadOnly-produced bundles). let v0 = crate::ids::SemVer::new(0, 1, 0); let mut m = Manifest::empty(DocumentId([1; 16])); m.profile_declarations = vec![profile(ProfileId::ReadOnly, v0, 1 << 20)]; let bundle = Bundle::create(MemStore::new(), FileUuid([1; 16]), m).unwrap(); assert_eq!(bundle.superblock().profile_id, ProfileId::ReadOnly); assert!(bundle.is_read_only()); // Round-trips: a reopen agrees it is read-only. let image = bundle.into_store().into_bytes(); assert!(Bundle::open(MemStore::from_bytes(image)) .unwrap() .is_read_only()); } #[test] fn editable_profile_is_preferred_for_the_active_superblock() { // [ReadOnly, Lite]: ReadOnly sorts first, but the bundle is emitted under // the editable Lite profile, so it is editable. let v0 = crate::ids::SemVer::new(0, 1, 0); let mut m = Manifest::empty(DocumentId([1; 16])); m.profile_declarations = vec![ profile(ProfileId::ReadOnly, v0, 1 << 20), profile(ProfileId::Lite, v0, 1 << 20), ]; let bundle = Bundle::create(MemStore::new(), FileUuid([1; 16]), m).unwrap(); assert_eq!(bundle.superblock().profile_id, ProfileId::Lite); assert!(!bundle.is_read_only()); } #[test] fn commit_validates_roots_under_the_profile_it_emits() { let make_bundle = |read_only_max, lite_max| { let v0 = crate::ids::SemVer::new(0, 1, 0); let mut m = Manifest::empty(DocumentId([1; 16])); m.profile_declarations = vec![ profile(ProfileId::ReadOnly, v0, read_only_max), profile(ProfileId::Lite, v0, lite_max), ]; Bundle::create(MemStore::new(), FileUuid([1; 16]), m).unwrap() }; let staged = || StagedChunk::operation_block(block::encode_block(&[vec![7; 16]])); let append_root = |ctx: &CommitContext| { let mut m = ctx.previous_manifest.clone(); m.operation_roots.push(ctx.new_chunks[0]); m }; // ReadOnly sorts first, but Lite is the emitted active profile. Its // smaller bound must reject this 24-byte operation block. let mut strict_lite = make_bundle(1024, 8); assert_eq!(strict_lite.superblock().profile_id, ProfileId::Lite); assert!(strict_lite.commit(&[staged()], append_root).is_err()); assert_eq!(strict_lite.generation(), 0); // Conversely, the selected Lite profile's larger bound must admit the // block even though canonical-first ReadOnly has a smaller bound. let mut permissive_lite = make_bundle(8, 1024); permissive_lite.commit(&[staged()], append_root).unwrap(); let root = permissive_lite.manifest().operation_roots[0]; assert!(permissive_lite.read_operation_block(&root).is_ok()); let reopened = Bundle::open(MemStore::from_bytes( permissive_lite.into_store().into_bytes(), )) .unwrap(); assert!(reopened.read_operation_block(&root).is_ok()); } /// Builds a minimal valid bundle image at `generation`, with the given /// profile declared and named by the superblock. fn craft_image(generation: u64, profile_id: ProfileId) -> Vec { let mut m = Manifest::empty(DocumentId([1; 16])); m.generation = generation; // Always declare Full (a distinct editable profile); add the named active // profile only when it is something other than Full, to avoid a duplicate. let mut decls = vec![ProfileDeclaration::full()]; if profile_id != ProfileId::Full { decls.push(profile( profile_id, crate::ids::SemVer::new(0, 1, 0), 1 << 20, )); } m.profile_declarations = decls; let payload = m.encode(); let mut image = vec![0u8; BODY_START as usize]; image.extend_from_slice(&payload); let sb = Superblock { generation, manifest_offset: BODY_START, manifest_length: payload.len() as u64, manifest_hash: manifest_chunk_hash(&payload), manifest_schema_version: SchemaVersion::V0, reduction_algorithm_version: ReductionAlgorithmVersion(0), profile_id, commit_state: CommitState::Committed, commit_timestamp: WallClockTime(0), }; image[0..crate::header::HEADER_LEN as usize] .copy_from_slice(&FixedHeader::new(FileUuid([1; 16])).encode()); image[SLOT_A_OFFSET as usize..SLOT_A_OFFSET as usize + SUPERBLOCK_LEN as usize] .copy_from_slice(&sb.encode()); image } #[test] fn read_only_profile_opens_read_only() { // Finding 6: a bundle whose active profile is ReadOnly opens read-only // (v0 does not auto-upgrade it). let image = craft_image(0, ProfileId::ReadOnly); let bundle = Bundle::open(MemStore::from_bytes(image)).unwrap(); assert!(bundle.is_read_only()); assert!(bundle.anomalies().is_empty()); // a normal read-only bundle } #[test] fn unsupported_custom_profile_opens_read_only_with_anomaly() { // Finding 2: a Custom (registry-defined) active profile is unsupported in // v0 — open read-only and surface it. let image = craft_image(0, ProfileId::Custom(crate::ids::ProfileRegistryId([9; 16]))); let bundle = Bundle::open(MemStore::from_bytes(image)).unwrap(); assert!(bundle.is_read_only()); assert!(bundle .anomalies() .contains(&IntegrityAnomaly::UnsupportedProfile)); } #[test] fn generation_exhaustion_is_an_error_not_a_panic() { // Finding 5: committing a generation-u64::MAX bundle returns an error. let image = craft_image(u64::MAX, ProfileId::Full); let mut bundle = Bundle::open(MemStore::from_bytes(image)).unwrap(); assert_eq!(bundle.generation(), u64::MAX); assert!(matches!( bundle.commit(&[], |ctx| ctx.previous_manifest.clone()), Err(BundleError::GenerationExhausted) )); } // ------------------------------------------------------------------ // Zstd read support (Chapter 8 §"Compression": reading zstd-compressed // chunks is a conformance MUST; this crate's writer still emits only // uncompressed chunks, so tests plant externally-compressed bytes). // ------------------------------------------------------------------ /// Appends externally-zstd-compressed bytes to the bundle body (as a /// foreign compressing writer would have) and returns the reference /// describing them. `declared_len` lets a test lie about the uncompressed /// length; honest callers pass `payload.len()`. fn plant_zstd_chunk( bundle: &mut Bundle, kind: ChunkKind, payload: &[u8], declared_len: u64, ) -> ChunkRef { let compressed = zstd::bulk::compress(payload, 3).unwrap(); let offset = bundle.write_cursor; bundle.store.write_at(offset, &compressed).unwrap(); bundle.write_cursor += compressed.len() as u64; let hash = content_hash_for(kind, SchemaVersion::V0, payload); ChunkRef { id: ChunkId(hash), kind, schema_version: SchemaVersion::V0, offset, compressed_length: compressed.len() as u64, uncompressed_length: declared_len, compression: CompressionAlgorithm::Zstd { level: 3 }, hash, } } /// Like [`plant_zstd_chunk`], but for a blob (bare `MUSCBLOB` addressing). fn plant_zstd_blob( bundle: &mut Bundle, payload: &[u8], declared_len: u64, ) -> BlobRef { let r = plant_zstd_chunk(bundle, ChunkKind::Blob, payload, declared_len); BlobRef { blob_id: BlobId(r.hash), media_type: "application/octet-stream".to_string(), offset: r.offset, compressed_length: r.compressed_length, uncompressed_length: declared_len, compression: r.compression, hash: r.hash, declared_max_uncompressed_length: None, } } #[test] fn zstd_compressed_chunk_round_trips_with_hash_verified() { // §Compression round-trip: an externally-compressed chunk reads back // byte-identical, and the content hash is verified over the // *uncompressed* bytes (the preimage rule: compression is metadata). let mut bundle = fresh_bundle(); let payload: Vec = b"layout-cache-bytes ".repeat(64); // compressible let r = plant_zstd_chunk( &mut bundle, ChunkKind::LayoutCache, &payload, payload.len() as u64, ); assert!( r.compressed_length < r.uncompressed_length, "fixture actually compressed" ); assert_eq!(bundle.read_chunk(&r).unwrap(), payload); // The hash check runs on the decompressed payload: a tampered declared // hash is caught even though the stored (compressed) bytes are intact. let mut tampered = r; tampered.hash = ContentHash([0; 32]); tampered.id = ChunkId(ContentHash([0; 32])); // keep id == hash assert!(matches!( bundle.read_chunk(&tampered), Err(BundleError::ChunkHashMismatch { .. }) )); } #[test] fn compressed_operation_root_commits_and_reopens() { // End to end: a compressed operation block can be published as a // canonical root (commit-time validation decompresses + verifies it), // survives a reopen, and streams back through read_operation_block. let mut bundle = fresh_bundle(); let envelopes = vec![b"env-1".to_vec(), b"env-2".to_vec()]; let payload = block::encode_block(&envelopes); let root = plant_zstd_chunk( &mut bundle, ChunkKind::OperationEnvelopeBlock, &payload, payload.len() as u64, ); bundle .commit(&[], |ctx| { let mut m = ctx.previous_manifest.clone(); m.operation_roots.push(root); m }) .unwrap(); let image = bundle.into_store().into_bytes(); let reopened = Bundle::open(MemStore::from_bytes(image)).unwrap(); reopened.verify_canonical_chunks().unwrap(); let stored_root = reopened.manifest().operation_roots[0]; assert_eq!( stored_root.compression, CompressionAlgorithm::Zstd { level: 3 } ); assert_eq!( reopened.read_operation_block(&stored_root).unwrap(), envelopes ); } #[test] fn zstd_stream_ending_short_of_declared_length_is_rejected() { // §Compression: "Decompression MUST verify the output length against // the declared uncompressed_length" — a stream that ends short of the // declaration is corruption, reported with both lengths. let mut bundle = fresh_bundle(); let payload = b"short-stream".to_vec(); let r = plant_zstd_chunk( &mut bundle, ChunkKind::LayoutCache, &payload, payload.len() as u64 + 5, // declares more than the stream yields ); assert!(matches!( bundle.read_chunk(&r), Err(BundleError::ChunkLengthMismatch { expected: 17, actual: 12, }) )); } #[test] fn zstd_stream_exceeding_declared_length_is_rejected() { // The dual failure: a stream that would decompress *past* the declared // length must be refused without allocating beyond the declaration // (the output buffer is sized by the declared length, so libzstd hits // destination-full and errors). let mut bundle = fresh_bundle(); let payload: Vec = b"overlong ".repeat(32); let r = plant_zstd_chunk( &mut bundle, ChunkKind::LayoutCache, &payload, payload.len() as u64 - 1, // declares less than the stream yields ); assert!(matches!( bundle.read_chunk(&r), Err(BundleError::Decompression(_)) )); } #[test] fn corrupt_or_truncated_zstd_stream_is_a_typed_error() { let mut bundle = fresh_bundle(); let payload: Vec = b"to-be-corrupted ".repeat(16); let r = plant_zstd_chunk( &mut bundle, ChunkKind::LayoutCache, &payload, payload.len() as u64, ); // Corrupt the frame header magic in the stored bytes: malformed stream. bundle.store.write_at(r.offset, &[0xFF]).unwrap(); assert!(matches!( bundle.read_chunk(&r), Err(BundleError::Decompression(_)) )); // Truncated stream: same compressed bytes, but the reference claims // fewer of them than the frame needs. let mut bundle = fresh_bundle(); let mut truncated = plant_zstd_chunk( &mut bundle, ChunkKind::LayoutCache, &payload, payload.len() as u64, ); truncated.compressed_length /= 2; assert!(matches!( bundle.read_chunk(&truncated), Err(BundleError::Decompression(_)) )); } #[test] fn zstd_compressed_blob_round_trips_and_fails_typed() { // The blob path shares the decode: round-trip, short-stream, corrupt. let mut bundle = fresh_bundle(); let payload: Vec = b"blob-audio-bytes ".repeat(64); let b = plant_zstd_blob(&mut bundle, &payload, payload.len() as u64); assert_eq!(bundle.read_blob(&b).unwrap(), payload); // Declared length beyond the stream's yield → typed error. let mut short = b.clone(); short.uncompressed_length = payload.len() as u64 + 3; assert!(matches!( bundle.read_blob(&short), Err(BundleError::ChunkLengthMismatch { .. }) )); // The declared_max cap still applies *before* decompression begins. let mut capped = b.clone(); capped.declared_max_uncompressed_length = Some(4); assert!(matches!( bundle.read_blob(&capped), Err(BundleError::ResourceLimitExceeded { .. }) )); // Corrupt stored stream → typed error, no panic. bundle.store.write_at(b.offset, &[0xFF]).unwrap(); assert!(matches!( bundle.read_blob(&b), Err(BundleError::Decompression(_)) )); } #[test] fn reserved_compression_is_still_unsupported() { // §Compression: Reserved algorithms belong to future format majors. let mut bundle = fresh_bundle(); op_block(&mut bundle, &[b"env"]); let mut r = bundle.manifest().operation_roots[0]; r.compression = CompressionAlgorithm::Reserved(7); assert!(matches!( bundle.read_chunk(&r), Err(BundleError::UnsupportedCompression) )); commit_blob(&mut bundle, b"blob"); let mut b = bundle.manifest().blob_roots[0].clone(); b.compression = CompressionAlgorithm::Reserved(7); assert!(matches!( bundle.read_blob(&b), Err(BundleError::UnsupportedCompression) )); } #[test] fn compressed_manifest_chunk_ref_is_rejected() { // §Manifest Encoding: the manifest chunk MUST be stored uncompressed in // this format version. A manifest reference declaring compression is // refused outright — even when the compressed bytes are a perfectly // valid zstd stream of a perfectly valid manifest. let mut bundle = fresh_bundle(); let manifest_payload = Manifest::empty(DocumentId([7; 16])).encode(); let r = plant_zstd_chunk( &mut bundle, ChunkKind::Manifest, &manifest_payload, manifest_payload.len() as u64, ); assert!(matches!( bundle.read_chunk(&r), Err(BundleError::CompressedManifest) )); } /// A minimal image whose superblock points at `stored` as the manifest /// payload, declaring `manifest_hash` for it. fn craft_image_with_manifest_bytes(stored: &[u8], manifest_hash: ContentHash) -> Vec { let mut image = vec![0u8; BODY_START as usize]; image.extend_from_slice(stored); let sb = Superblock { generation: 0, manifest_offset: BODY_START, manifest_length: stored.len() as u64, manifest_hash, manifest_schema_version: SchemaVersion::V0, reduction_algorithm_version: ReductionAlgorithmVersion(0), profile_id: ProfileId::Full, commit_state: CommitState::Committed, commit_timestamp: WallClockTime(0), }; image[0..crate::header::HEADER_LEN as usize] .copy_from_slice(&FixedHeader::new(FileUuid([1; 16])).encode()); image[SLOT_A_OFFSET as usize..SLOT_A_OFFSET as usize + SUPERBLOCK_LEN as usize] .copy_from_slice(&sb.encode()); image } #[test] fn open_rejects_a_bundle_whose_manifest_bytes_are_compressed() { // §Manifest Encoding: "Implementations MUST reject as malformed any // bundle whose manifest payload is not directly parseable as a // canonical manifest chunk's uncompressed bytes." The superblock // carries no compression field, so the stored bytes are treated as the // payload — a compressed manifest fails with a typed error either way // a hostile writer declares its hash. let payload = Manifest::empty(DocumentId([1; 16])).encode(); let compressed = zstd::bulk::compress(&payload, 3).unwrap(); // (a) Hash declared over the true (uncompressed) manifest content: // the stored bytes fail hash verification → no valid superblock. let image = craft_image_with_manifest_bytes(&compressed, manifest_chunk_hash(&payload)); assert!(matches!( Bundle::open(MemStore::from_bytes(image)), Err(BundleError::NoValidSuperblock) )); // (b) Colluding hash over the compressed bytes: the slot verifies, but // the payload is not parseable as a manifest → rejected as // malformed. let image = craft_image_with_manifest_bytes(&compressed, manifest_chunk_hash(&compressed)); assert!(matches!( Bundle::open(MemStore::from_bytes(image)), Err(BundleError::Decode(_)) )); } }