From c590664e974f0a31e03c62c1716795cddb26a9fe Mon Sep 17 00:00:00 2001 From: forhappy Date: Wed, 7 Oct 2026 02:04:21 -0700 Subject: [PATCH 001/102] perf: share Cell publication and authorize follower append windows Keep per-Cell fenced root selection while uploading bounded application-scoped capture cohorts. Issue signed append windows through fresh enrollment and serialize native proof release with durable closure. Document the remaining bundle-coverage protocol and preserve qualification failures. Bundle-based bucket acknowledgments remain disabled. --- crates/cellule-axum/examples/fleet/server.rs | 88 +++- crates/cellule-axum/examples/fleet/tests.rs | 26 + .../cellule-axum/examples/fleet/transport.rs | 171 ++++-- crates/cellule-axum/examples/fleet/wire.rs | 13 +- .../cellule-axum/examples/sql_metrics/mod.rs | 48 +- .../examples/sql_metrics/storage.rs | 8 +- crates/cellule-ltx/api-prelude.txt | 4 + crates/cellule-ltx/docs/packed-root-format.md | 51 +- crates/cellule-ltx/src/cell_layout.rs | 6 + crates/cellule-ltx/src/lib.rs | 3 +- crates/cellule-ltx/src/replica/cache.rs | 40 +- .../src/replica/compaction/scratch.rs | 22 +- .../src/replica/compaction/source.rs | 15 +- crates/cellule-ltx/src/replica/mod.rs | 11 + crates/cellule-ltx/src/replica/packed.rs | 30 +- crates/cellule-ltx/src/replica/prepare.rs | 21 +- crates/cellule-ltx/src/replica/root.rs | 33 +- crates/cellule-ltx/src/replica/shared/mod.rs | 464 +++++++++++++++++ crates/cellule-ltx/src/replica/upload.rs | 7 +- crates/cellule-ltx/src/replica/verify.rs | 23 +- crates/cellule-ltx/tests/cell/roots.rs | 1 + crates/cellule-ltx/tests/cell/roots/packed.rs | 2 +- crates/cellule-ltx/tests/cell/roots/shared.rs | 268 ++++++++++ crates/cellule-runtime/docs/append-grants.md | 72 +++ .../docs/failover-and-followers.md | 8 + crates/cellule-runtime/model/README.md | 15 + crates/cellule-runtime/model/WriteProofs.cfg | 4 + crates/cellule-runtime/model/WriteProofs.tla | 100 ++++ .../model/WriteProofsNoFence.cfg | 4 + .../model/WriteProofsNodeOnly.cfg | 4 + .../model/WriteProofsWallOnly.cfg | 4 + crates/cellule-runtime/model/check.sh | 13 +- .../cellule-runtime/src/cell/actor/acquire.rs | 1 + .../src/cell/actor/requests.rs | 3 + .../cellule-runtime/src/cell/actor/runtime.rs | 10 +- .../cellule-runtime/src/cell/actor/state.rs | 3 +- crates/cellule-runtime/src/fleet/telemetry.rs | 36 ++ .../cellule-runtime/src/follower/grant/mod.rs | 247 +++++++++ .../src/follower/grant/tests.rs | 486 ++++++++++++++++++ crates/cellule-runtime/src/follower/mod.rs | 87 +++- .../src/follower/records/append.rs | 8 + .../src/node/append_grant/mod.rs | 264 ++++++++++ .../src/node/append_grant/tests.rs | 23 + .../cellule-runtime/src/node/directory/mod.rs | 11 + crates/cellule-runtime/src/node/mod.rs | 1 + crates/cellule-runtime/src/publication/mod.rs | 147 +++++- .../src/publication/shared/mod.rs | 305 +++++++++++ .../src/publication/shared/tests.rs | 256 +++++++++ .../src/recovery/retention/mod.rs | 3 + .../src/recovery/retention/tests.rs | 147 ++++++ docs/bundle-coverage-proof.md | 84 +++ docs/write-performance-delivery.md | 56 +- scripts/perf/README.md | 5 +- scripts/perf/build.py | 9 +- scripts/perf/report.py | 10 +- scripts/perf/run.py | 9 +- 56 files changed, 3652 insertions(+), 138 deletions(-) create mode 100644 crates/cellule-ltx/src/replica/shared/mod.rs create mode 100644 crates/cellule-ltx/tests/cell/roots/shared.rs create mode 100644 crates/cellule-runtime/docs/append-grants.md create mode 100644 crates/cellule-runtime/model/WriteProofs.cfg create mode 100644 crates/cellule-runtime/model/WriteProofs.tla create mode 100644 crates/cellule-runtime/model/WriteProofsNoFence.cfg create mode 100644 crates/cellule-runtime/model/WriteProofsNodeOnly.cfg create mode 100644 crates/cellule-runtime/model/WriteProofsWallOnly.cfg create mode 100644 crates/cellule-runtime/src/follower/grant/mod.rs create mode 100644 crates/cellule-runtime/src/follower/grant/tests.rs create mode 100644 crates/cellule-runtime/src/node/append_grant/mod.rs create mode 100644 crates/cellule-runtime/src/node/append_grant/tests.rs create mode 100644 crates/cellule-runtime/src/publication/shared/mod.rs create mode 100644 crates/cellule-runtime/src/publication/shared/tests.rs create mode 100644 docs/bundle-coverage-proof.md diff --git a/crates/cellule-axum/examples/fleet/server.rs b/crates/cellule-axum/examples/fleet/server.rs index 98121504..e973c3c0 100644 --- a/crates/cellule-axum/examples/fleet/server.rs +++ b/crates/cellule-axum/examples/fleet/server.rs @@ -1,4 +1,4 @@ -//! Private follower process: mTLS, fresh enrollment and canonical fsynced storage. +//! Private follower process: signed grants, mTLS and canonical fsynced storage. use super::*; use axum::{ Extension, Router, @@ -9,7 +9,9 @@ use axum::{ }; use bytes::Bytes; use cellule_peer_http::PeerTlsIdentity; -use cellule_runtime::follower::FollowerStore; +use cellule_runtime::follower::{ + AppendGrantIssuer, AppendGrantPeer, FollowerStore, GrantedFollowerAppend, +}; use ed25519_dalek::VerifyingKey; use prost::Message; use tokio::sync::{OwnedSemaphorePermit, Semaphore}; @@ -41,6 +43,7 @@ pub(super) async fn serve( cellule_ltx::Limits::default(), cellule_ltx::DiskBudget::new(1 << 30), )? + .with_append_grant_receiver(session) .with_telemetry( cellule_runtime::fleet::telemetry::CellTelemetryHandle::from_sink(metrics.clone()), ); @@ -165,21 +168,30 @@ async fn handle_inner(server: Server, peer: PeerTlsIdentity, encoded: Bytes) -> if request.member != server.member.as_bytes() { return Err(Error::PeerAuthorization("capacity log member differs")); } - let enrolled = server - .metrics - .enrollment( - true, + let enrolled = if matches!(request.operation, 2..=4) { + Some( server - .directory - .peer_verifier(sender, peer.certificate(), peer.public_key(), clock()?), + .metrics + .enrollment( + true, + server.directory.peer_verifier( + sender, + peer.certificate(), + peer.public_key(), + clock()?, + ), + ) + .await?, ) - .await; - let enrolled = enrolled?; + } else { + None + }; // Directory I/O consumed time. Preserve the accepted horizon across wall // clock rollback, while monotonic elapsed time still expires the request. let now = wire::request_time(started_ms, clock()?, started_at.elapsed())?; validate_request(&request, now, "after directory verification")?; server.lease.check()?; + let deadline = request.deadline_ms; let mut reply = wire::Reply { member: server.member.as_bytes().to_vec(), request_digest: blake3::hash(&body).as_bytes().to_vec(), @@ -190,20 +202,21 @@ async fn handle_inner(server: Server, peer: PeerTlsIdentity, encoded: Bytes) -> if sender != leader { return Err(Error::PeerAuthorization("capacity append sender differs")); } - enrolled.authorize_log_append( - server.member, - request.epoch, - request.covered_through, - clock()?, - )?; let phase = std::time::Instant::now(); let result = server .store - .append( - leader, - request.epoch, - request.frames.into_iter().map(Bytes::from).collect(), - request.covered_through, + .append_granted( + GrantedFollowerAppend { + peer: AppendGrantPeer { + session: sender, + certificate: peer.certificate(), + public_key: peer.public_key(), + }, + log_epoch: request.epoch, + grant: Digest::try_from(request.grant.as_slice())?, + frames: request.frames.into_iter().map(Bytes::from).collect(), + }, + server.lease.clone(), ) .await; server @@ -213,6 +226,33 @@ async fn handle_inner(server: Server, peer: PeerTlsIdentity, encoded: Bytes) -> reply.base_sequence = receipt.base_sequence; reply.durable_through = receipt.durable_through; } + 5 => { + if sender != leader { + return Err(Error::PeerAuthorization("capacity grant sender differs")); + } + let grant = server + .metrics + .enrollment( + true, + server.store.open_append_grant( + AppendGrantPeer { + session: sender, + certificate: peer.certificate(), + public_key: peer.public_key(), + }, + request.epoch, + request.first_sequence, + AppendGrantIssuer { + directory: &server.directory, + signing_key: server.tls.signing_key(), + lease: &server.lease, + now_ms: now, + }, + ), + ) + .await?; + reply.grant = grant.encode(); + } 3 => { if sender != leader { return Err(Error::PeerAuthorization("capacity retire sender differs")); @@ -262,7 +302,13 @@ async fn handle_inner(server: Server, peer: PeerTlsIdentity, encoded: Bytes) -> } _ => return Err(Error::PeerAuthorization("capacity operation differs")), } + drop(enrolled); server.lease.check()?; + if wire::request_time(started_ms, clock()?, started_at.elapsed())? >= deadline { + return Err(Error::PeerAuthorization( + "capacity reply exceeded request horizon", + )); + } let reply = wire::sign( reply.encode_to_vec(), server.tls.signing_key(), diff --git a/crates/cellule-axum/examples/fleet/tests.rs b/crates/cellule-axum/examples/fleet/tests.rs index ef4be314..59be10cc 100644 --- a/crates/cellule-axum/examples/fleet/tests.rs +++ b/crates/cellule-axum/examples/fleet/tests.rs @@ -12,6 +12,7 @@ fn follower_signature_rejects_modified_payload_wrong_key_and_wrong_direction() { leader: vec![1; 16], epoch: 1, operation: 1, + grant: vec![7; 32], frames: vec![vec![3; 128]], deadline_ms: 2000, ..Default::default() @@ -49,6 +50,7 @@ fn stale_or_oversized_append_and_frames_on_retirement_are_refused() { leader: vec![1; 16], epoch: 1, operation: 1, + grant: vec![7; 32], frames: vec![vec![3; 128]], deadline_ms: 2000, ..Default::default() @@ -70,6 +72,7 @@ fn stale_or_oversized_append_and_frames_on_retirement_are_refused() { request.operation = 3; assert!(wire::validate(&request, 1000).is_err()); request.frames.clear(); + request.grant.clear(); assert!(wire::validate(&request, 1000).is_ok()); } @@ -81,6 +84,7 @@ fn directory_verification_clock_rollback_preserves_horizon_without_extending_exp leader: vec![1; 16], epoch: 1, operation: 1, + grant: vec![7; 32], frames: vec![vec![3; 128]], deadline_ms: 11_000, ..Default::default() @@ -100,3 +104,25 @@ fn directory_verification_clock_rollback_preserves_horizon_without_extending_exp let now = wire::request_time(1000, 11_000, Duration::from_millis(25)).unwrap(); assert!(wire::validate(&request, now).is_err()); } + +#[test] +fn grant_issuance_and_append_tokens_have_separate_signed_shapes() { + let mut request = wire::Request { + sender: vec![1; 16], + leader: vec![1; 16], + member: vec![2; 16], + epoch: 1, + operation: 5, + first_sequence: 1, + deadline_ms: 2000, + ..Default::default() + }; + assert!(wire::validate(&request, 1000).is_ok()); + request.grant = vec![7; 32]; + assert!(wire::validate(&request, 1000).is_err()); + request.operation = 1; + request.frames = vec![vec![3; 128]]; + assert!(wire::validate(&request, 1000).is_ok()); + request.grant.pop(); + assert!(wire::validate(&request, 1000).is_err()); +} diff --git a/crates/cellule-axum/examples/fleet/transport.rs b/crates/cellule-axum/examples/fleet/transport.rs index a8715bff..74a419a6 100644 --- a/crates/cellule-axum/examples/fleet/transport.rs +++ b/crates/cellule-axum/examples/fleet/transport.rs @@ -10,9 +10,15 @@ use futures_util::{StreamExt, future::BoxFuture}; use prost::Message; use std::collections::HashMap; +struct ClientGrant { + authorization: cellule_runtime::node::append_grant::NodeAppendGrant, + expires: std::time::Instant, +} + struct Member { original: NodeAdvertisement, client: reqwest::Client, + grant: tokio::sync::Mutex>, } pub(super) struct Transport { @@ -38,7 +44,14 @@ impl Transport { .client(original.certificate(), original.verifying_key()?.to_bytes()) .map_err(transport_error)?; if enrolled - .insert(original.node(), Member { original, client }) + .insert( + original.node(), + Member { + original, + client, + grant: tokio::sync::Mutex::new(None), + }, + ) .is_some() { return Err(Error::Node("duplicate capacity follower")); @@ -53,31 +66,48 @@ impl Transport { }) } - async fn round_trip(&self, member: NodeId, mut request: wire::Request) -> Result { + async fn round_trip( + &self, + member: NodeId, + request: &mut wire::Request, + ) -> Result<(wire::Reply, NodeAdvertisement)> { let peer = self .members .get(&member) .ok_or(Error::PeerAuthorization("unknown capacity follower"))?; - let fresh = self - .metrics - .enrollment( - false, - self.directory - .load_if_live(peer.original.session(), clock()?), - ) - .await; - let fresh = fresh?.ok_or(Error::Fenced)?; - let actual = fresh.advertisement(); - if actual.node() != member - || actual.certificate() != peer.original.certificate() - || actual.endpoint() != peer.original.endpoint() - || actual.verifying_key()? != peer.original.verifying_key()? - { - return Err(Error::Fenced); - } + let actual = if request.operation == 1 { + // The installed signed grant pins this original receiver boot. + // Only renewal reads directory enrollment; ordinary frames still + // use pinned mTLS and exact signed request/response bytes. + peer.original.clone() + } else { + let fresh = self + .metrics + .enrollment( + false, + self.directory + .load_if_live(peer.original.session(), clock()?), + ) + .await; + let fresh = fresh?.ok_or(Error::Fenced)?; + let actual = fresh.advertisement(); + if actual.node() != member + || actual.certificate() != peer.original.certificate() + || actual.endpoint() != peer.original.endpoint() + || actual.verifying_key()? != peer.original.verifying_key()? + { + return Err(Error::Fenced); + } + actual.clone() + }; request.sender = self.sender.as_bytes().to_vec(); request.member = member.as_bytes().to_vec(); - request.deadline_ms = clock()?.checked_add(10_000).ok_or(Error::Deadline)?; + let deadline = clock()?.checked_add(10_000).ok_or(Error::Deadline)?; + request.deadline_ms = if request.deadline_ms == 0 { + deadline + } else { + request.deadline_ms.min(deadline) + }; let phase = std::time::Instant::now(); let body = request.encode_to_vec(); let digest = blake3::hash(&body); @@ -101,9 +131,7 @@ impl Transport { .send() .await .map_err(transport_error)?; - if !response.status().is_success() { - return Err(Error::Peer("capacity follower HTTP rejection")); - } + let response = response.error_for_status().map_err(transport_error)?; let mut stream = response.bytes_stream(); let mut bytes = Vec::new(); while let Some(part) = stream.next().await { @@ -132,7 +160,7 @@ impl Transport { } self.metrics .peer_phase(PeerPhase::ReplyVerify, phase.elapsed()); - Ok(reply) + Ok((reply, actual)) } } @@ -145,7 +173,7 @@ fn request(operation: u32, leader: SessionId, epoch: u64) -> wire::Request { } } -fn receipt(reply: wire::Reply) -> FollowerReceipt { +fn receipt((reply, _): (wire::Reply, NodeAdvertisement)) -> FollowerReceipt { FollowerReceipt { base_sequence: reply.base_sequence, durable_through: reply.durable_through, @@ -159,14 +187,97 @@ impl NodeLogTransport for Transport { append: AppendRequest, ) -> BoxFuture<'a, Result> { Box::pin(async move { + let peer = self + .members + .get(&member) + .ok_or(Error::PeerAuthorization("unknown capacity follower"))?; + // Renewal cannot invalidate another append in flight on this member. + let mut cached = peer.grant.lock().await; + let first = append + .frames + .first() + .ok_or(Error::Peer("empty capacity append"))?; + let last = append + .frames + .last() + .ok_or(Error::Peer("empty capacity append"))?; + let first = + cellule_ltx::inspect_node_frame(first.clone(), cellule_ltx::Limits::default())? + .scope() + .node_sequence; + let last = + cellule_ltx::inspect_node_frame(last.clone(), cellule_ltx::Limits::default())? + .scope() + .node_sequence; let mut message = request(1, append.leader_session, append.log_epoch); message.frames = append .frames .into_iter() .map(|frame| frame.to_vec()) .collect(); - message.covered_through = append.covered_through; - self.round_trip(member, message).await.map(receipt) + for attempt in 0..2 { + let valid = cached.as_ref().is_some_and(|grant| { + std::time::Instant::now() < grant.expires + && grant + .authorization + .authorize( + self.sender, + self.tls.certificate(), + self.tls.signing_key().verifying_key().to_bytes(), + append.log_epoch, + first, + last, + ) + .is_ok() + }); + if !valid { + *cached = None; + let started = std::time::Instant::now(); + let mut issue = request(5, append.leader_session, append.log_epoch); + issue.first_sequence = first; + let (reply, receiver) = self.round_trip(member, &mut issue).await?; + let authorization = + cellule_runtime::node::append_grant::NodeAppendGrant::verify( + &reply.grant, + &receiver, + clock()?, + )?; + authorization.authorize( + self.sender, + self.tls.certificate(), + self.tls.signing_key().verifying_key().to_bytes(), + append.log_epoch, + first, + last, + )?; + let lifetime = authorization + .expires_at_ms() + .checked_sub(authorization.issued_at_ms()) + .ok_or(Error::Deadline)?; + let lifetime = u64::try_from(lifetime).map_err(|_| Error::Deadline)?; + *cached = Some(ClientGrant { + authorization, + expires: started + Duration::from_millis(lifetime), + }); + } + let grant = cached.as_ref().ok_or(Error::Fenced)?; + message.grant = grant.authorization.digest().as_bytes().to_vec(); + message.deadline_ms = grant.authorization.expires_at_ms(); + let result = self.round_trip(member, &mut message).await.map(receipt); + match result { + Ok(receipt) => return Ok(receipt), + Err(error) => { + *cached = None; + if attempt == 1 { + return Err(error); + } + // A lost/expired first response may already have fsynced. + // Renew and replay the same frames once; native duplicate + // digests reconcile it without rerunning a command. + } + } + } + Err(Error::Peer("capacity append retry bound")) }) } @@ -176,7 +287,7 @@ impl NodeLogTransport for Transport { seal: SealRequest, ) -> BoxFuture<'a, Result> { Box::pin(async move { - self.round_trip(member, request(2, seal.leader_session, seal.log_epoch)) + self.round_trip(member, &mut request(2, seal.leader_session, seal.log_epoch)) .await .map(receipt) }) @@ -190,7 +301,7 @@ impl NodeLogTransport for Transport { Box::pin(async move { let mut message = request(3, retire.leader_session, retire.log_epoch); message.covered_through = retire.covered_through; - self.round_trip(member, message).await.map(receipt) + self.round_trip(member, &mut message).await.map(receipt) }) } @@ -212,7 +323,7 @@ impl NodeLogTransport for Transport { Box::pin(async move { let mut message = request(4, tail.leader_session, tail.log_epoch); message.first_sequence = tail.first_sequence; - let reply = self.round_trip(member, message).await?; + let (reply, _) = self.round_trip(member, &mut message).await?; Ok(FollowerTailPage { frames: reply.frames.into_iter().map(Bytes::from).collect(), next_sequence: reply.next_sequence, diff --git a/crates/cellule-axum/examples/fleet/wire.rs b/crates/cellule-axum/examples/fleet/wire.rs index fb903cba..c580018d 100644 --- a/crates/cellule-axum/examples/fleet/wire.rs +++ b/crates/cellule-axum/examples/fleet/wire.rs @@ -4,8 +4,8 @@ use ed25519_dalek::{Signature, Signer, SigningKey, Verifier, VerifyingKey}; use prost::Message; use std::time::Duration; -pub(super) const REQUEST_DOMAIN: &[u8] = b"cellule.axum-capacity.node-log.request.v1\0"; -pub(super) const RESPONSE_DOMAIN: &[u8] = b"cellule.axum-capacity.node-log.response.v1\0"; +pub(super) const REQUEST_DOMAIN: &[u8] = b"cellule.axum-capacity.node-log.request.v2\0"; +pub(super) const RESPONSE_DOMAIN: &[u8] = b"cellule.axum-capacity.node-log.response.v2\0"; pub(super) const MAX_REQUEST_BYTES: usize = (65 << 20) + 65_536; pub(super) const MAX_RESPONSE_BYTES: usize = 2 << 20; pub(super) const PATH: &str = "/internal/capacity/node-log"; @@ -38,6 +38,8 @@ pub(super) struct Request { pub first_sequence: u64, #[prost(int64, tag = "9")] pub deadline_ms: i64, + #[prost(bytes = "vec", tag = "10")] + pub grant: Vec, } #[derive(Clone, PartialEq, Message)] @@ -56,6 +58,8 @@ pub(super) struct Reply { pub frames: Vec>, #[prost(uint64, optional, tag = "7")] pub next_sequence: Option, + #[prost(bytes = "vec", tag = "8")] + pub grant: Vec, } pub(super) fn sign(body: Vec, key: &SigningKey, domain: &[u8]) -> Vec { @@ -83,9 +87,12 @@ pub(super) fn validate(request: &Request, now: i64) -> Result<()> { || request.member.len() != 16 || request.leader.len() != 16 || request.epoch == 0 - || !(1..=4).contains(&request.operation) + || !(1..=5).contains(&request.operation) || (request.operation == 1 && (request.frames.is_empty() || request.frames.len() > 64)) || (request.operation != 1 && !request.frames.is_empty()) + || (request.operation == 1 && request.grant.len() != 32) + || (request.operation != 1 && !request.grant.is_empty()) + || (request.operation == 5 && request.first_sequence == 0) { return Err(Error::PeerAuthorization("invalid capacity log request")); } diff --git a/crates/cellule-axum/examples/sql_metrics/mod.rs b/crates/cellule-axum/examples/sql_metrics/mod.rs index e0dd426e..430c1fed 100644 --- a/crates/cellule-axum/examples/sql_metrics/mod.rs +++ b/crates/cellule-axum/examples/sql_metrics/mod.rs @@ -1,7 +1,7 @@ //! Finite, nonblocking runtime and provider measurements for this application. use cellule_runtime::fleet::telemetry::{ CellTelemetry, CommandResponseSource, DurabilitySubmissionOutcome, PrimitiveOperationKind, - PrimitiveOperationOutcome, PublicationTiming, + PrimitiveOperationOutcome, PublicationTiming, SharedPublicationTiming, }; use cellule_store::{StorageObservation, StorageObserver, StorageOperation, StorageOutcome}; use std::{ @@ -39,6 +39,7 @@ pub(super) struct QueryMetrics { log_append_bytes: AtomicU64, selected_roots: AtomicU64, materialized_commits: AtomicU64, + shared: SharedMetrics, follower_frames: AtomicU64, follower_sync_calls: AtomicU64, follower_failures: AtomicU64, @@ -46,6 +47,19 @@ pub(super) struct QueryMetrics { sample_session: std::sync::OnceLock, } +#[derive(Default)] +struct SharedMetrics { + cohorts: AtomicU64, + failures: AtomicU64, + cells: AtomicU64, + rows: AtomicU64, + bytes: AtomicU64, + pressure: AtomicU64, + large: AtomicU64, + queue: Histogram, + upload: Histogram, +} + // Application-owned instrumentation: fixed histograms, 100-us upper // bounds through two seconds. No request or Cell labels and no actor locks. const WRITE_BUCKET_US: u64 = 100; @@ -292,6 +306,25 @@ impl CellTelemetry for QueryMetrics { self.writes.objects.fetch_add(objects, Ordering::Relaxed); self.writes.bytes.fetch_add(bytes, Ordering::Relaxed); } + fn shared_publication(&self, timing: SharedPublicationTiming) { + self.shared.cohorts.fetch_add(1, Ordering::Relaxed); + self.shared + .failures + .fetch_add(u64::from(!timing.succeeded), Ordering::Relaxed); + self.shared.cells.fetch_add(timing.cells, Ordering::Relaxed); + self.shared.rows.fetch_add(timing.rows, Ordering::Relaxed); + self.shared.bytes.fetch_add(timing.bytes, Ordering::Relaxed); + self.shared.queue.observe(timing.queue); + self.shared.upload.observe(timing.upload); + } + fn shared_publication_fallback(&self, pressure: bool) { + if pressure { + &self.shared.pressure + } else { + &self.shared.large + } + .fetch_add(1, Ordering::Relaxed); + } fn ltx_phase(&self, phase: cellule_ltx::LtxPhase, elapsed: Duration, _succeeded: bool) { use cellule_ltx::LtxPhase; match phase { @@ -360,6 +393,8 @@ impl QueryMetrics { ("publication", &self.writes.publication), ("compaction", &self.writes.compaction), ("dirty_admission", &self.writes.dirty_admission), + ("shared_queue", &self.shared.queue), + ("shared_upload", &self.shared.upload), ] { histograms.insert(label.into(), histogram.raw()); } @@ -472,6 +507,17 @@ impl QueryMetrics { "data_sync_calls": self.follower_sync_calls.load(Ordering::Relaxed), "failures": self.follower_failures.load(Ordering::Relaxed) }, + "shared_publication": { + "cohorts": self.shared.cohorts.load(Ordering::Relaxed), + "failures": self.shared.failures.load(Ordering::Relaxed), + "cells": self.shared.cells.load(Ordering::Relaxed), + "rows": self.shared.rows.load(Ordering::Relaxed), + "bytes": self.shared.bytes.load(Ordering::Relaxed), + "pressure_fallbacks": self.shared.pressure.load(Ordering::Relaxed), + "large_fallbacks": self.shared.large.load(Ordering::Relaxed), + "queue": self.shared.queue.snapshot(), + "upload": self.shared.upload.snapshot() + }, "storage_operations": storage, "storage_families": storage_families, "peer_phases": PeerPhase::ALL.iter().map(|phase| (phase.label().to_owned(), self.peer[*phase as usize].snapshot())).collect::>(), diff --git a/crates/cellule-axum/examples/sql_metrics/storage.rs b/crates/cellule-axum/examples/sql_metrics/storage.rs index fb30336f..aba60c1b 100644 --- a/crates/cellule-axum/examples/sql_metrics/storage.rs +++ b/crates/cellule-axum/examples/sql_metrics/storage.rs @@ -94,9 +94,11 @@ impl Accounting { ) -> Arc { let index = location.map_or(3, |path| { let name = path.as_ref(); - if [".ltx", ".index", ".dir", ".root", ".bundle", ".pack"] - .iter() - .any(|suffix| name.ends_with(suffix)) + if [ + ".ltx", ".index", ".dir", ".root", ".bundle", ".pack", ".spack", + ] + .iter() + .any(|suffix| name.ends_with(suffix)) { 0 } else if name diff --git a/crates/cellule-ltx/api-prelude.txt b/crates/cellule-ltx/api-prelude.txt index 2c3fb4aa..afe4a796 100644 --- a/crates/cellule-ltx/api-prelude.txt +++ b/crates/cellule-ltx/api-prelude.txt @@ -42,8 +42,12 @@ RootPreparation RootPreparationFuture RootPreparationMetadata RootRef +SHARED_PUBLICATION_BYTES +SHARED_PUBLICATION_ROWS ScratchMonitor SegmentInfo +SharedAppend +SharedCaptures TransactionError VerifiedNodeFrame VerifiedPlan diff --git a/crates/cellule-ltx/docs/packed-root-format.md b/crates/cellule-ltx/docs/packed-root-format.md index bfbd85e5..efd6b3c5 100644 --- a/crates/cellule-ltx/docs/packed-root-format.md +++ b/crates/cellule-ltx/docs/packed-root-format.md @@ -1,7 +1,7 @@ # Bounded packed dependencies -Cell root JSON version 2 is the current development format. It replaces version -1 atomically; there is no legacy decoder or dual-write path. Native LTX bytes, +Cell root JSON version 3 is the current development format. It replaces versions +1 and 2 atomically; there is no legacy decoder or dual-write path. Native LTX bytes, node frame formats, Cell fencing, accumulating lineage and authority selection remain their existing contracts. @@ -36,6 +36,47 @@ Small compaction outputs use the same representation. Sparse reads continue verifying individual frame hashes from the authenticated directory. Compaction authenticates the complete packed source, including bytes outside page frames. +## Shared publication object + +An application-wide `.spack` holds at most 64 rows and 256 KiB. It uses +`CRBSH001`, a big-endian u32 row count and four reserved zero bytes (16-byte +header), followed by contiguous rows. Each row is a 32-byte Cell ID and 16-byte +incarnation, then an ordinary `CRBPACK1` header, native LTX bytes and index. +Objects live beneath `shared/objects/.spack` in the application prefix. +Application-owned writer, reader, recovery and backup credentials must allow +that shared prefix as well as the existing Cell object prefixes. + +Root descriptors require `shared: true` and `packed: true`, pin the complete +object digest and exact body/index extents, and retain the native body digest. +Ordinary descriptors require `shared: false`. Inventory, compaction and full +restore authenticate the complete object and the selected row's Cell scope. +Sparse reads authenticate its scoped header and individual page frame hashes; +verified header entries consume the existing bounded metadata cache. Identical +native bytes in two Cells cannot substitute for the authenticated row scope. + +The runtime acquires one of 64 cohort credits before Cell dirty admission. +It probes and releases root capacity before the actor freezes its retained +range; no dirty permit waits for a shared uploader. +It charges pinned files, index and ownership tables before preparing inputs; +coalescing and the file-backed upload use the existing host and disk budgets. +The dispatcher bounds in-flight uploads to eight, under the existing host +I/O/scratch budgets. A cohort combines already-prepared rows and freezes when the queue is idle, +or at 256 KiB, 64 rows or a 1-ms assembly bound. Cells sharing a cohort +must have the same provider identity and application prefix. Large cuts, +compaction, or insufficient retained-memory admission use ordinary preparation. +Cancelled waiters leave dispatched upload and scratch cleanup owned by the lane. +The lane joins after all Cell actors and publishers during shutdown. + +Upload grants no durability authority. Each Cell independently prepares its +root, accumulates lineage and performs its existing fenced control CAS. Failed +siblings cannot select another Cell's root. Collection includes shared objects +in its complete application mark set, including dormant roots and backup pins, +and still requires exclusive maintenance and a grace boundary. + +For `c` commands per cohort and `r` commands per selected Cell root, ordinary +publication still needs approximately `1/c + 3/r` PUTs per command. Sharing +payloads cannot remove the per-Cell authority floor or prove bucket parity. + ## Inline directory leaf The root has a canonical `directory_inline` field containing lowercase hex or @@ -58,6 +99,10 @@ writes and cancellation. Large objects keep separate bounded streams. ## Evidence and cutover +`tests/cell/roots/shared.rs` covers exact independent restore, identical-byte +cross-Cell substitution, malformed rows and shared compaction. Runtime tests +cover cancelled waiters, minimum budgets and dormant-sibling collection. + `tests/cell/roots/packed.rs` validates the header and original bytes, corrupts header/body/index bytes, truncates and deletes the object, and rejects cross-Cell scope and invalid extents. Coalescing, compaction, sparse activation, restore, @@ -68,6 +113,6 @@ Runtime lineage and fenced control selection add two successful PUTs. Follow the workspace [development format policy](../../cellule-runtime/docs/storage.md#format-policy). All producers and consumers must deploy the current format together, including backup, recovery and collection workers. Qualification uses fresh isolated -prefixes. An older binary cannot read version 2 roots; rollback requires a +prefixes. An older binary cannot read version 3 roots; rollback requires a verified logical export/rebuild or recreating disposable development fixtures. Do not run mixed root-format binaries against one prefix. diff --git a/crates/cellule-ltx/src/cell_layout.rs b/crates/cellule-ltx/src/cell_layout.rs index bf50feb5..18752bfc 100644 --- a/crates/cellule-ltx/src/cell_layout.rs +++ b/crates/cellule-ltx/src/cell_layout.rs @@ -27,6 +27,8 @@ pub enum CellObjectKind { Bundle, /// Bounded segment and authenticated fixed-width index in one object. Packed, + /// Application-scoped, bounded captures shared by multiple Cell roots. + SharedPacked, } impl CellObjectKind { @@ -38,6 +40,7 @@ impl CellObjectKind { Self::Root => "root", Self::Bundle => "bundle", Self::Packed => "pack", + Self::SharedPacked => "spack", } } } @@ -177,6 +180,9 @@ impl CellStorageLayout { digest: &[u8; 32], kind: CellObjectKind, ) -> Path { + if kind == CellObjectKind::SharedPacked { + return self.application_path(&format!("shared/objects/{}.spack", encode_hex(digest))); + } self.application_path(&format!( "cells/{}/inc/{}/objects/{}.{}", encode_hex(cell), diff --git a/crates/cellule-ltx/src/lib.rs b/crates/cellule-ltx/src/lib.rs index 5339faef..a3791778 100644 --- a/crates/cellule-ltx/src/lib.rs +++ b/crates/cellule-ltx/src/lib.rs @@ -82,7 +82,8 @@ pub use node_frame::{ pub use replica::{ CellPagedDatabase, CellReplica, CellWritableDatabase, PreparedRoot, PublicationCost, ReadOnlyRoot, RecoveryOverlay, RootObjectRef, RootPreparation, RootPreparationFuture, - RootPreparationMetadata, RootRef, VerifiedRoot, + RootPreparationMetadata, RootRef, SHARED_PUBLICATION_BYTES, SHARED_PUBLICATION_ROWS, + SharedAppend, SharedCaptures, VerifiedRoot, }; #[cfg(feature = "replica")] pub use writable_vfs::Hydration; diff --git a/crates/cellule-ltx/src/replica/cache.rs b/crates/cellule-ltx/src/replica/cache.rs index 4d4d643c..9171e984 100644 --- a/crates/cellule-ltx/src/replica/cache.rs +++ b/crates/cellule-ltx/src/replica/cache.rs @@ -31,15 +31,15 @@ impl Cache { if let Some(existing) = self.objects.get(&key) { return Arc::clone(existing); } - while self.bytes.saturating_add(bytes.len()) > CACHE_BYTES { + while self.bytes.saturating_add(bytes.len().max(1024)) > CACHE_BYTES { let Some(old) = self.order.pop_front() else { break; }; if let Some(removed) = self.objects.remove(&old) { - self.bytes -= removed.len(); + self.bytes -= removed.len().max(1024); } } - self.bytes += bytes.len(); + self.bytes += bytes.len().max(1024); self.order.push_back(key.clone()); self.objects.insert(key, Arc::clone(&bytes)); bytes @@ -101,6 +101,40 @@ pub(super) fn insert( .insert(key(layout, cell, incarnation, digest, kind), bytes)) } +pub(super) fn shared_header( + layout: &CellStorageLayout, + object: [u8; 32], + offset: u64, + expected: &[u8], + verified: bool, +) -> Result { + // Reuse the metadata band and its cap. The minimum entry charge bounds + // small-header key/table overhead as well as retained payload bytes. + let key = Key { + store: layout.immutable_cache_identity(), + path: format!( + "{}@{offset}", + layout.incarnation_object_path( + &[0; 32], + &[0; 16], + &object, + CellObjectKind::SharedPacked + ) + ), + digest: *blake3::hash(expected).as_bytes(), + }; + let mut cache = band(CellObjectKind::Directory)? + .lock() + .map_err(|_| LtxError::InvalidState("immutable object cache poisoned"))?; + if verified { + cache.insert(key, expected.into()); + return Ok(true); + } + Ok(cache + .get(&key) + .is_some_and(|bytes| bytes.as_ref() == expected)) +} + #[cfg(test)] mod tests { use super::*; diff --git a/crates/cellule-ltx/src/replica/compaction/scratch.rs b/crates/cellule-ltx/src/replica/compaction/scratch.rs index 46564fc7..bf5c1cee 100644 --- a/crates/cellule-ltx/src/replica/compaction/scratch.rs +++ b/crates/cellule-ltx/src/replica/compaction/scratch.rs @@ -6,7 +6,7 @@ use cellule_store::{MultipartUploadSource, StorageError}; use super::CellReplica; use crate::{CellObjectKind, Host, LtxError, Result, environment::FileIo}; -pub(super) async fn upload( +pub(in crate::replica) async fn upload( replica: &CellReplica, scratch: &Arc, source: &Path, @@ -81,16 +81,17 @@ fn storage_read_error(error: LtxError) -> StorageError { } } -pub(super) struct ScratchFiles { - pub(super) host: Host, +pub(in crate::replica) struct ScratchFiles { + pub(in crate::replica) host: Host, runtime: tokio::runtime::Handle, cleaned: Option>, directory: PathBuf, paths: Vec, + disk: Option, } impl ScratchFiles { - pub(super) fn new( + pub(in crate::replica) fn new( host: Host, directory: &Path, runtime: tokio::runtime::Handle, @@ -102,10 +103,19 @@ impl ScratchFiles { cleaned: Some(cleaned), directory: directory.to_owned(), paths: Vec::new(), + disk: None, } } - pub(super) fn create(&mut self, label: &str) -> Result { + pub(in crate::replica) fn with_disk_reservation( + mut self, + disk: crate::DiskReservation, + ) -> Self { + self.disk = Some(disk); + self + } + + pub(in crate::replica) fn create(&mut self, label: &str) -> Result { static NEXT: std::sync::atomic::AtomicU64 = std::sync::atomic::AtomicU64::new(1); for _ in 0..16 { let path = self.directory.join(format!( @@ -178,6 +188,7 @@ impl Drop for ScratchFiles { let paths = std::mem::take(&mut self.paths); let host = self.host.clone(); let cleaned = self.cleaned.take(); + let disk = self.disk.take(); // Keep dirty/recovery/scratch admission until the last file and queued // job have finished, then remove files through the same job ceiling. self.runtime.spawn(async move { @@ -190,6 +201,7 @@ impl Drop for ScratchFiles { }) .await; drop(host); + drop(disk); if let Some(cleaned) = cleaned { let _ = cleaned.send(()); } diff --git a/crates/cellule-ltx/src/replica/compaction/source.rs b/crates/cellule-ltx/src/replica/compaction/source.rs index a9928f6e..551295af 100644 --- a/crates/cellule-ltx/src/replica/compaction/source.rs +++ b/crates/cellule-ltx/src/replica/compaction/source.rs @@ -29,7 +29,10 @@ pub(super) async fn spool_selected( let results = stream::iter(planned.into_iter().enumerate().map( |(order, (descriptor, body_start, index_start))| { async move { - if descriptor.object_kind() == CellObjectKind::Packed { + if matches!( + descriptor.object_kind(), + CellObjectKind::Packed | CellObjectKind::SharedPacked + ) { return spool_packed( replica, descriptor, @@ -90,7 +93,7 @@ async fn spool_packed( &replica.cell, &replica.incarnation, &descriptor.object_digest(), - CellObjectKind::Packed, + descriptor.object_kind(), ); let permit = replica.host.io_permit().await?; let (bytes, _) = replica @@ -98,11 +101,15 @@ async fn spool_packed( .store() .get_with_etag_bounded(&path, super::super::upload::SINGLE_PUT_BYTES) .await?; - super::super::packed::verify(&bytes, &descriptor)?; + if descriptor.object_kind() == CellObjectKind::SharedPacked { + super::super::shared::verify(&bytes, &descriptor, &replica.cell, &replica.incarnation)?; + } else { + super::super::packed::verify(&bytes, &descriptor)?; + } let body_offset = descriptor.offset() as usize; let index_offset = (descriptor.offset() + descriptor.info.size_bytes) as usize; let body = bytes.slice(body_offset..index_offset); - let index = bytes.slice(index_offset..); + let index = bytes.slice(index_offset..index_offset + descriptor.index_length as usize); let mut body_file = scratch.open(body_destination).await?; let mut index_file = scratch.open(index_destination).await?; replica diff --git a/crates/cellule-ltx/src/replica/mod.rs b/crates/cellule-ltx/src/replica/mod.rs index 42020d9e..af48e280 100644 --- a/crates/cellule-ltx/src/replica/mod.rs +++ b/crates/cellule-ltx/src/replica/mod.rs @@ -19,6 +19,8 @@ mod compaction; pub(crate) mod directory; mod merge; mod packed; +mod shared; +pub use shared::{SHARED_PUBLICATION_BYTES, SHARED_PUBLICATION_ROWS, SharedAppend, SharedCaptures}; mod preparation; mod prepare; pub use preparation::{RootPreparation, RootPreparationFuture, RootPreparationMetadata}; @@ -330,6 +332,7 @@ pub struct CellPagedDatabase { directory_height: u32, directory_inline: Option>, extents: Arc>, + shared_segments: Arc>, page_size: u32, database_pages: u32, position: Position, @@ -495,6 +498,8 @@ impl CellPagedDatabase { .offset .checked_add(u64::from(entry.length)) .ok_or(LtxError::LTXCorrupted)?; + self.verify_shared_range(entry.object, entry.offset, end, origin) + .await?; let path = self.replica.layout.incarnation_object_path( &self.replica.cell, &self.replica.incarnation, @@ -617,6 +622,8 @@ impl CellPagedDatabase { span: DirectorySpan, origin: crate::LtxReadOrigin, ) -> Result { + self.verify_shared_range(span.object, span.start, span.end, origin) + .await?; let extent = self .extents .get(&span.object) @@ -943,19 +950,23 @@ struct AppendInput { body: AppendBody, } +#[derive(Clone)] enum AppendBody { Native(Arc), Frozen(Bytes), Packed(Arc), Bundle, + SharedUploaded, } #[derive(Clone, Copy)] enum BodyLocation { Native, Bundle { digest: [u8; 32], offset: u64 }, + Shared { digest: [u8; 32], offset: u64 }, } +#[derive(Clone)] struct PreparedSegment { descriptor: SegmentDescriptor, index: Bytes, diff --git a/crates/cellule-ltx/src/replica/packed.rs b/crates/cellule-ltx/src/replica/packed.rs index e826cfbb..e7d4bff4 100644 --- a/crates/cellule-ltx/src/replica/packed.rs +++ b/crates/cellule-ltx/src/replica/packed.rs @@ -19,7 +19,12 @@ pub(super) async fn freeze( .checked_add(segment.descriptor.info.size_bytes) .and_then(|n| n.checked_add(segment.index.len() as u64)) .ok_or(LtxError::LTXCorrupted)?; - if length > SINGLE_PUT_BYTES || matches!(segment.body, AppendBody::Bundle) { + if length > SINGLE_PUT_BYTES + || matches!( + segment.body, + AppendBody::Bundle | AppendBody::SharedUploaded + ) + { return Ok(segment); } let body = match &segment.body { @@ -29,7 +34,9 @@ pub(super) async fn freeze( replica.host.run(move || source.read_small(size)).await?? } AppendBody::Frozen(bytes) => bytes.clone(), - AppendBody::Bundle | AppendBody::Packed(_) => return Err(LtxError::LTXCorrupted), + AppendBody::Bundle | AppendBody::Packed(_) | AppendBody::SharedUploaded => { + return Err(LtxError::LTXCorrupted); + } }; if body.len() as u64 != segment.descriptor.info.size_bytes || *blake3::hash(&body).as_bytes() != segment.descriptor.info.blake3 @@ -40,11 +47,7 @@ pub(super) async fn freeze( // Freeze before provider I/O; retained memory is bounded by the existing // single-PUT limit and survives cancellation with its preparation owner. let mut bytes = BytesMut::with_capacity(length as usize); - bytes.extend_from_slice(MAGIC); - bytes.extend_from_slice(&(body.len() as u64).to_be_bytes()); - bytes.extend_from_slice(&(segment.index.len() as u64).to_be_bytes()); - bytes.extend_from_slice(&[0; 8]); - bytes.extend_from_slice(&segment.descriptor.info.blake3); + bytes.extend_from_slice(&header(&segment.descriptor)); bytes.extend_from_slice(&body); bytes.extend_from_slice(&segment.index); let bytes = bytes.freeze(); @@ -59,7 +62,9 @@ pub(super) async fn freeze( let source: Arc = match segment.body { AppendBody::Native(source) => source, AppendBody::Frozen(bytes) => Arc::new(super::upload::FrozenCapture(bytes)), - AppendBody::Bundle | AppendBody::Packed(_) => return Err(LtxError::LTXCorrupted), + AppendBody::Bundle | AppendBody::Packed(_) | AppendBody::SharedUploaded => { + return Err(LtxError::LTXCorrupted); + } }; // Retain the pinned source and shared index, rather than all frozen bodies // across a preparation cohort. Upload buffers only while holding host I/O. @@ -105,6 +110,15 @@ pub(super) fn verify(bytes: &Bytes, descriptor: &SegmentDescriptor) -> Result<() Ok(()) } +pub(super) fn header(descriptor: &SegmentDescriptor) -> [u8; 64] { + let mut header = [0; 64]; + header[..8].copy_from_slice(MAGIC); + header[8..16].copy_from_slice(&descriptor.info.size_bytes.to_be_bytes()); + header[16..24].copy_from_slice(&descriptor.index_length.to_be_bytes()); + header[32..64].copy_from_slice(&descriptor.info.blake3); + header +} + struct PackedCapture { header: Bytes, source: Arc, diff --git a/crates/cellule-ltx/src/replica/prepare.rs b/crates/cellule-ltx/src/replica/prepare.rs index 5bdb394f..11039704 100644 --- a/crates/cellule-ltx/src/replica/prepare.rs +++ b/crates/cellule-ltx/src/replica/prepare.rs @@ -126,7 +126,7 @@ impl CellReplica { .await } - fn admit_capture_batch(&self, cuts: &CaptureBatch) -> Result<()> { + pub(super) fn admit_capture_batch(&self, cuts: &CaptureBatch) -> Result<()> { if cuts.segments.is_empty() { return Err(LtxError::InvalidState("empty Cell append")); } @@ -147,7 +147,7 @@ impl CellReplica { Ok(()) } - async fn prepare_captured_inputs( + pub(super) async fn prepare_captured_inputs( &self, segments: &[crate::LocalSegment], ) -> Result> { @@ -585,6 +585,13 @@ impl CellReplica { digest, offset, ), + BodyLocation::Shared { digest, offset } => SegmentDescriptor::shared( + input.info, + *blake3::hash(&input.index).as_bytes(), + input.index.len() as u64, + digest, + offset, + ), }; Ok(PreparedSegment { descriptor, @@ -602,7 +609,11 @@ impl CellReplica { // Admit every original cut before reducing its representation. A merge // cannot rescue a gap, invalid endpoint, or over-budget original chain. self.validate_chain(&descriptors, target)?; - if bundle.is_none() { + if bundle.is_none() + && prepared + .iter() + .all(|segment| !matches!(segment.body, AppendBody::SharedUploaded)) + { prepared = coalesce::run(self, prepared).await?; prepared = stream::iter( prepared @@ -872,14 +883,14 @@ impl CellReplica { }) } - fn validate_metadata(&self, commit_sequence: u64, schema: u32) -> Result<()> { + pub(super) fn validate_metadata(&self, commit_sequence: u64, schema: u32) -> Result<()> { if schema == 0 || commit_sequence > i64::MAX as u64 { return Err(LtxError::InvalidState("invalid Cell root metadata")); } Ok(()) } - fn validate_append_sequence( + pub(super) fn validate_append_sequence( &self, base: &Option, commit_sequence: u64, diff --git a/crates/cellule-ltx/src/replica/root.rs b/crates/cellule-ltx/src/replica/root.rs index 42a1fb2c..437e6642 100644 --- a/crates/cellule-ltx/src/replica/root.rs +++ b/crates/cellule-ltx/src/replica/root.rs @@ -34,6 +34,7 @@ pub(super) struct SegmentDescriptor { length: u64, level: u8, packed: bool, + shared: bool, } impl SegmentDescriptor { @@ -49,6 +50,7 @@ impl SegmentDescriptor { length, level: 0, packed: false, + shared: false, } } @@ -69,6 +71,7 @@ impl SegmentDescriptor { length, level: 0, packed: false, + shared: false, } } @@ -88,14 +91,28 @@ impl SegmentDescriptor { length, level: 0, packed: true, + shared: false, } } + pub(super) fn shared( + info: SegmentInfo, + index_digest: [u8; 32], + index_length: u64, + object_digest: [u8; 32], + offset: u64, + ) -> Self { + let mut descriptor = Self::bundled(info, index_digest, index_length, object_digest, offset); + descriptor.packed = true; + descriptor.shared = true; + descriptor + } + pub(super) fn index_extent(&self) -> ([u8; 32], crate::CellObjectKind, u64) { if self.packed { ( self.object_digest, - crate::CellObjectKind::Packed, + self.object_kind(), self.offset + self.length, ) } else { @@ -126,7 +143,9 @@ impl SegmentDescriptor { } pub(super) fn object_kind(&self) -> crate::CellObjectKind { - if self.packed { + if self.shared { + crate::CellObjectKind::SharedPacked + } else if self.packed { crate::CellObjectKind::Packed } else if self.object_digest == self.info.blake3 { crate::CellObjectKind::Ltx @@ -157,9 +176,10 @@ impl SegmentDescriptor { || self.index_length > u64::from(info.database_pages) * 60 || self.length != info.size_bytes || self.offset.checked_add(self.length).is_none() + || (self.shared && (!self.packed || self.offset < super::shared::FIRST_BODY_OFFSET)) || (self.object_kind() == crate::CellObjectKind::Ltx && self.offset != 0) || (self.packed - && (self.offset != super::packed::HEADER_BYTES + && ((!self.shared && self.offset != super::packed::HEADER_BYTES) || self .offset .checked_add(self.length) @@ -222,6 +242,7 @@ struct SegmentWire { page_size: u32, post_checksum: String, pre_checksum: String, + shared: bool, size_bytes: String, } @@ -285,7 +306,7 @@ pub(super) fn encode_root(root: &RootDocument) -> Result> { .collect(), segments: root.segments.iter().map(segment_wire).collect(), txid: root.txid.to_string(), - version: 2, + version: 3, })?; if bytes.len() as u64 > ROOT_BYTES { return Err(LtxError::Limit(crate::LimitKind::CellRootBytes)); @@ -308,7 +329,7 @@ pub(super) fn decode_root(bytes: &[u8]) -> Result { return Err(LtxError::Limit(crate::LimitKind::CellRootBytes)); } let wire: RootWire = serde_json::from_slice(bytes)?; - if wire.version != 2 { + if wire.version != 3 { return Err(LtxError::LTXCorrupted); } let root = RootDocument { @@ -379,6 +400,7 @@ fn segment_wire(segment: &SegmentDescriptor) -> SegmentWire { min_txid: segment.info.min_txid.to_string(), object_digest: encode_hex(&segment.object_digest), packed: segment.packed, + shared: segment.shared, offset: segment.offset.to_string(), page_size: segment.info.page_size, post_checksum: checksum(segment.info.post_checksum), @@ -406,6 +428,7 @@ fn segment_descriptor(segment: SegmentWire) -> Result { length: decimal(&segment.length)?, level: segment.level, packed: segment.packed, + shared: segment.shared, }) } diff --git a/crates/cellule-ltx/src/replica/shared/mod.rs b/crates/cellule-ltx/src/replica/shared/mod.rs new file mode 100644 index 00000000..ba8a6a8b --- /dev/null +++ b/crates/cellule-ltx/src/replica/shared/mod.rs @@ -0,0 +1,464 @@ +//! Bounded application-scoped capture sharing. Upload supplies no authority. + +use std::{collections::BTreeSet, path::Path, sync::Arc}; + +use bytes::Bytes; + +use super::{ + AppendBaseState, AppendBody, AppendInput, BodyLocation, CellPagedDatabase, CellReplica, + PreparedRoot, PreparedSegment, RootRef, SegmentDescriptor, coalesce, compaction::scratch, + packed, upload::SINGLE_PUT_BYTES, +}; +use crate::{CaptureBatch, CellObjectKind, LtxError, Position, Result}; + +const MAGIC: &[u8; 8] = b"CRBSH001"; +const HEADER_BYTES: u64 = 16; +const SCOPE_BYTES: u64 = 48; +/// Bound shared data to one verified small-object transfer. +pub const SHARED_PUBLICATION_BYTES: u64 = SINGLE_PUT_BYTES; +/// Bound cohort ownership tables independently of compressed body size. +pub const SHARED_PUBLICATION_ROWS: usize = 64; +pub(super) const FIRST_BODY_OFFSET: u64 = HEADER_BYTES + SCOPE_BYTES + packed::HEADER_BYTES; + +/// Verified, pinned captures awaiting one bounded shared upload. +/// +/// Construction admits every original segment before representation reduction. +/// This owns no Cell preparation permit and confers no root or writer authority. +pub struct SharedCaptures { + replica: CellReplica, + position: Position, + segments: Vec, + original: Vec, + encoded_bytes: u64, +} + +impl CellPagedDatabase { + pub(super) async fn verify_shared_range( + &self, + object: [u8; 32], + start: u64, + end: u64, + origin: crate::LtxReadOrigin, + ) -> Result<()> { + for descriptor in self.shared_segments.iter().filter(|descriptor| { + descriptor.object_digest() == object + && descriptor.offset() < end + && descriptor.offset() + descriptor.info.size_bytes > start + }) { + let mut expected = [0; (SCOPE_BYTES + packed::HEADER_BYTES) as usize]; + expected[..32].copy_from_slice(&self.replica.cell); + expected[32..48].copy_from_slice(&self.replica.incarnation); + expected[48..].copy_from_slice(&packed::header(descriptor)); + let offset = descriptor + .offset() + .checked_sub(expected.len() as u64) + .ok_or(LtxError::LTXCorrupted)?; + if super::cache::shared_header(&self.replica.layout, object, offset, &expected, false)? + { + continue; + } + let path = self.replica.layout.incarnation_object_path( + &self.replica.cell, + &self.replica.incarnation, + &object, + CellObjectKind::SharedPacked, + ); + let _permit = self.replica.host.io_permit().await?; + let bytes = self + .replica + .layout + .store() + .range_get(&path, offset..descriptor.offset()) + .await; + self.replica.host.observe_ltx_origin_request( + origin, + bytes.is_ok(), + bytes.as_ref().map_or(0, Bytes::len), + ); + if bytes?.as_ref() != expected { + return Err(LtxError::LTXCorrupted); + } + super::cache::shared_header(&self.replica.layout, object, offset, &expected, true)?; + } + Ok(()) + } +} + +impl SharedCaptures { + /// Exact encoded row bytes, excluding the one shared header. + #[must_use] + pub const fn encoded_bytes(&self) -> u64 { + self.encoded_bytes + } + + /// Number of scoped rows retained by this input. + #[must_use] + pub fn rows(&self) -> usize { + self.segments.len() + } + + /// Immutable namespace and transport binding; supplies no authority. + #[must_use] + pub fn storage_binding(&self) -> (u64, String) { + ( + self.replica.layout.immutable_cache_identity(), + self.replica.layout.application_prefix().to_string(), + ) + } +} + +/// One Cell's verified extents in an uploaded shared object. +/// +/// The private factory waits for exact upload completion. The runtime must still +/// select a prepared root under that Cell's writer fence before acknowledging. +#[derive(Clone)] +pub struct SharedAppend { + replica: CellReplica, + position: Position, + segments: Vec, + original: Vec, +} + +impl CellReplica { + /// Pins and verifies captures eligible for one bounded shared publication. + /// Large captures retain the ordinary streaming preparation path. + pub async fn shared_captures(&self, cuts: &CaptureBatch) -> Result> { + self.admit_capture_batch(cuts)?; + let upper = cuts + .segments + .iter() + .try_fold(HEADER_BYTES, |bytes, segment| { + bytes + .checked_add(SCOPE_BYTES + packed::HEADER_BYTES) + .and_then(|bytes| bytes.checked_add(segment.info().size_bytes)) + .and_then(|bytes| { + bytes.checked_add(u64::from(segment.info().database_pages) * 60) + }) + .ok_or(LtxError::Limit(crate::LimitKind::CellBundleBytes)) + })?; + if cuts.segments.len() > SHARED_PUBLICATION_ROWS || upper > SINGLE_PUT_BYTES { + return Ok(None); + } + let inputs = self.prepare_captured_inputs(&cuts.segments).await?; + let segments: Vec<_> = inputs + .into_iter() + .map(|input| PreparedSegment { + descriptor: SegmentDescriptor::native( + input.info, + *blake3::hash(&input.index).as_bytes(), + input.index.len() as u64, + ), + index: input.index, + body: input.body, + }) + .collect(); + let original = segments + .iter() + .map(|segment: &PreparedSegment| segment.descriptor.clone()) + .collect(); + let segments = coalesce::run(self, segments).await?; + let encoded_bytes = segments.iter().try_fold(0_u64, |bytes, segment| { + bytes + .checked_add(SCOPE_BYTES + packed::HEADER_BYTES) + .and_then(|bytes| bytes.checked_add(segment.descriptor.info.size_bytes)) + .and_then(|bytes| bytes.checked_add(segment.descriptor.index_length)) + .ok_or(LtxError::LTXCorrupted) + })?; + Ok(Some(SharedCaptures { + replica: self.clone(), + position: cuts.position, + segments, + original, + encoded_bytes, + })) + } + + /// Uploads one file-backed cohort once and returns independently scoped inputs. + /// The caller owns retained-memory admission for row/index tables and the + /// bounded upload buffer. Scratch uses the same host ledger as compaction. + pub async fn upload_shared( + inputs: Vec, + scratch_directory: &Path, + ) -> Result> { + let first = inputs + .first() + .ok_or(LtxError::InvalidState("empty shared publication"))?; + let binding = first.storage_binding(); + let replica = first.replica.clone(); + let size = inputs.iter().try_fold(HEADER_BYTES, |size, input| { + if input.storage_binding() != binding { + return Err(LtxError::InvalidState( + "shared publication crosses storage binding", + )); + } + size.checked_add(input.encoded_bytes) + .ok_or(LtxError::LTXCorrupted) + })?; + let rows: usize = inputs.iter().map(SharedCaptures::rows).sum(); + if size > SINGLE_PUT_BYTES || rows == 0 || rows > SHARED_PUBLICATION_ROWS { + return Err(LtxError::Limit(crate::LimitKind::CellBundleBytes)); + } + let host = replica.host.for_scratch(size).await?; + let disk = host.reserve_local_disk(size)?; + let (cleaned, cleanup) = tokio::sync::oneshot::channel(); + let runtime = tokio::runtime::Handle::current(); + let directory = scratch_directory.to_owned(); + let work_host = host.clone(); + let built = host + .run(move || -> Result<_> { + let mut scratch = + scratch::ScratchFiles::new(work_host, &directory, runtime, cleaned) + .with_disk_reservation(disk); + let destination = scratch.create("shared-publication")?; + let work = Arc::new(scratch); + let mut file = work.host.filesystem.open_rw(&destination)?; + let mut hasher = blake3::Hasher::new(); + let mut offset = 0_u64; + let mut write = |bytes: &[u8]| -> Result<()> { + file.write_all(bytes)?; + hasher.update(bytes); + offset += bytes.len() as u64; + Ok(()) + }; + let mut header = [0; HEADER_BYTES as usize]; + header[..8].copy_from_slice(MAGIC); + header[8..12].copy_from_slice(&(rows as u32).to_be_bytes()); + write(&header)?; + let mut appends = Vec::with_capacity(inputs.len()); + let mut identities = BTreeSet::new(); + let mut body_offset = HEADER_BYTES; + for input in inputs { + let mut segments = Vec::with_capacity(input.segments.len()); + for mut segment in input.segments { + let info = segment.descriptor.info.clone(); + if !identities.insert(( + input.replica.cell, + input.replica.incarnation, + info.min_txid, + info.max_txid, + )) { + return Err(LtxError::InvalidState("duplicate shared capture range")); + } + let body = match &segment.body { + AppendBody::Native(source) => source.read_small(info.size_bytes)?, + AppendBody::Frozen(bytes) => bytes.clone(), + _ => { + return Err(LtxError::InvalidState( + "shared input is not a native capture", + )); + } + }; + if body.len() as u64 != info.size_bytes + || *blake3::hash(&body).as_bytes() != info.blake3 + { + return Err(LtxError::ChecksumMismatch); + } + write(&input.replica.cell)?; + write(&input.replica.incarnation)?; + write(&packed::header(&segment.descriptor))?; + body_offset += SCOPE_BYTES + packed::HEADER_BYTES; + write(&body)?; + write(&segment.index)?; + segment.descriptor = SegmentDescriptor::shared( + info.clone(), + segment.descriptor.index_digest, + segment.descriptor.index_length, + [0; 32], + body_offset, + ); + body_offset += info.size_bytes + segment.index.len() as u64; + segment.body = AppendBody::SharedUploaded; + segments.push(segment); + } + appends.push(SharedAppend { + replica: input.replica, + position: input.position, + segments, + original: input.original, + }); + } + if offset != size || file.file_len()? != size { + return Err(LtxError::LTXCorrupted); + } + file.sync_all()?; + let digest = *hasher.finalize().as_bytes(); + for append in &mut appends { + for segment in &mut append.segments { + segment.descriptor = SegmentDescriptor::shared( + segment.descriptor.info.clone(), + segment.descriptor.index_digest, + segment.descriptor.index_length, + digest, + segment.descriptor.offset(), + ); + } + } + Ok((appends, digest, work, destination)) + }) + .await; + let result = match built { + Ok(Ok((appends, digest, scratch, path))) => scratch::upload( + &replica, + &scratch, + &path, + size, + &digest, + CellObjectKind::SharedPacked, + ) + .await + .map(|()| appends), + Ok(Err(error)) | Err(error) => Err(error), + }; + drop(host); + let _ = cleanup.await; + result + } + + /// Prepares a scoped shared append through the canonical root factory. + pub async fn prepare_shared( + &self, + base: Option<&RootRef>, + append: &SharedAppend, + commit_sequence: u64, + schema: u32, + ) -> Result { + if self.scope() != append.replica.scope() + || self.layout.immutable_cache_identity() + != append.replica.layout.immutable_cache_identity() + || self.layout.application_prefix() != append.replica.layout.application_prefix() + { + return Err(LtxError::InvalidState( + "shared append belongs to another Cell or store", + )); + } + self.validate_metadata(commit_sequence, schema)?; + let mut replica = self.clone(); + replica.host = self.host.for_dirty().await?; + let graph = match base { + Some(root) => Some(replica.load_graph(root).await?), + None => None, + }; + replica.validate_append_sequence(&graph, commit_sequence)?; + // Check the original chain and budget before reduced representations. + // Coalescing cannot hide a missing cut or rescue an over-budget append. + let mut original = graph + .as_ref() + .map(|graph| graph.descriptors.clone()) + .unwrap_or_default(); + original.extend(append.original.iter().cloned()); + replica.validate_chain(&original, append.position)?; + let inputs = append + .segments + .iter() + .map(|segment| AppendInput { + info: segment.descriptor.info.clone(), + location: BodyLocation::Shared { + digest: segment.descriptor.object_digest(), + offset: segment.descriptor.offset(), + }, + index: segment.index.clone(), + body: AppendBody::SharedUploaded, + }) + .collect(); + replica + .prepare_append( + base, + graph.map(AppendBaseState::from), + inputs, + append.position, + commit_sequence, + schema, + None, + ) + .await + } +} + +pub(super) fn verify( + bytes: &Bytes, + descriptor: &SegmentDescriptor, + cell: &[u8; 32], + incarnation: &[u8; 16], +) -> Result<()> { + if bytes.len() < HEADER_BYTES as usize + || bytes.len() as u64 > SINGLE_PUT_BYTES + || &bytes[..8] != MAGIC + || bytes[12..16] != [0; 4] + { + return Err(LtxError::LTXCorrupted); + } + if *blake3::hash(bytes).as_bytes() != descriptor.object_digest() { + return Err(LtxError::ChecksumMismatch); + } + let rows = u32::from_be_bytes( + bytes[8..12] + .try_into() + .map_err(|_| LtxError::LTXCorrupted)?, + ) as usize; + if rows == 0 || rows > SHARED_PUBLICATION_ROWS { + return Err(LtxError::LTXCorrupted); + } + let mut offset = HEADER_BYTES as usize; + let mut found = false; + let mut identities = BTreeSet::new(); + for _ in 0..rows { + let start = offset + .checked_add(SCOPE_BYTES as usize) + .ok_or(LtxError::LTXCorrupted)?; + let body = start + .checked_add(packed::HEADER_BYTES as usize) + .ok_or(LtxError::LTXCorrupted)?; + let header = bytes.get(start..body).ok_or(LtxError::LTXCorrupted)?; + if &header[..8] != b"CRBPACK1" || header[24..32] != [0; 8] { + return Err(LtxError::LTXCorrupted); + } + let length = u64::from_be_bytes( + header[8..16] + .try_into() + .map_err(|_| LtxError::LTXCorrupted)?, + ); + let index = u64::from_be_bytes( + header[16..24] + .try_into() + .map_err(|_| LtxError::LTXCorrupted)?, + ); + let end = (body as u64) + .checked_add(length) + .and_then(|end| end.checked_add(index)) + .filter(|end| *end <= bytes.len() as u64) + .ok_or(LtxError::LTXCorrupted)? as usize; + if length < 128 + || index == 0 + || !index.is_multiple_of(crate::paged::ENTRY_BYTES as u64) + || !identities.insert((bytes.slice(offset..start), bytes.slice(start + 32..body))) + { + return Err(LtxError::LTXCorrupted); + } + if blake3::hash(&bytes[body..body + length as usize]) + .as_bytes() + .as_slice() + != &header[32..64] + { + return Err(LtxError::ChecksumMismatch); + } + if body as u64 == descriptor.offset() { + if &bytes[offset..offset + 32] != cell || &bytes[offset + 32..start] != incarnation { + return Err(LtxError::LTXCorrupted); + } + let record = bytes.slice(start..end); + let local = SegmentDescriptor::packed( + descriptor.info.clone(), + descriptor.index_digest, + descriptor.index_length, + *blake3::hash(&record).as_bytes(), + ); + packed::verify(&record, &local)?; + found = true; + } + offset = end; + } + if offset != bytes.len() || !found { + return Err(LtxError::LTXCorrupted); + } + Ok(()) +} diff --git a/crates/cellule-ltx/src/replica/upload.rs b/crates/cellule-ltx/src/replica/upload.rs index ebf383ca..90def2e3 100644 --- a/crates/cellule-ltx/src/replica/upload.rs +++ b/crates/cellule-ltx/src/replica/upload.rs @@ -67,6 +67,11 @@ impl CellReplica { index, body, } = segment; + if matches!(body, AppendBody::SharedUploaded) { + // Only the verified shared-upload factory constructs this input. + // Both body and index already belong to its one immutable object. + return Ok(()); + } if let AppendBody::Packed(source) = body { let path = self.layout.incarnation_object_path( &self.cell, @@ -88,7 +93,7 @@ impl CellReplica { let source: Arc = match body { AppendBody::Native(source) => source, AppendBody::Frozen(bytes) => Arc::new(FrozenCapture(bytes)), - AppendBody::Bundle | AppendBody::Packed(_) => { + AppendBody::Bundle | AppendBody::Packed(_) | AppendBody::SharedUploaded => { return Err(LtxError::InvalidState("native Cell body source missing")); } }; diff --git a/crates/cellule-ltx/src/replica/verify.rs b/crates/cellule-ltx/src/replica/verify.rs index 295e295d..4ace5c28 100644 --- a/crates/cellule-ltx/src/replica/verify.rs +++ b/crates/cellule-ltx/src/replica/verify.rs @@ -88,12 +88,15 @@ impl CellReplica { }), ); for descriptor in &graph.descriptors { - if descriptor.object_kind() == CellObjectKind::Packed { + if matches!( + descriptor.object_kind(), + CellObjectKind::Packed | CellObjectKind::SharedPacked + ) { let path = self.layout.incarnation_object_path( &self.cell, &self.incarnation, &descriptor.object_digest(), - CellObjectKind::Packed, + descriptor.object_kind(), ); let _permit = self.host.io_permit().await?; let result = self @@ -107,10 +110,14 @@ impl CellReplica { result.as_ref().map_or(0, |(bytes, _)| bytes.len()), ); let (bytes, _) = result?; - packed::verify(&bytes, descriptor)?; + if descriptor.object_kind() == CellObjectKind::SharedPacked { + shared::verify(&bytes, descriptor, &self.cell, &self.incarnation)?; + } else { + packed::verify(&bytes, descriptor)?; + } objects.insert(RootObjectRef { digest: descriptor.object_digest(), - kind: CellObjectKind::Packed, + kind: descriptor.object_kind(), }); continue; } @@ -465,6 +472,14 @@ impl VerifiedRoot { directory_height: document.directory_height, directory_inline: document.directory_inline.clone(), extents: Arc::new(extents), + shared_segments: Arc::new( + descriptors + .into_iter() + .filter(|descriptor| { + descriptor.object_kind() == CellObjectKind::SharedPacked + }) + .collect(), + ), page_size: document.page_size, database_pages: document.database_pages, position: root.position, diff --git a/crates/cellule-ltx/tests/cell/roots.rs b/crates/cellule-ltx/tests/cell/roots.rs index a27fcf23..81c9a48f 100644 --- a/crates/cellule-ltx/tests/cell/roots.rs +++ b/crates/cellule-ltx/tests/cell/roots.rs @@ -40,4 +40,5 @@ mod lifecycle; mod packed; mod preparation; mod prepare_cost; +mod shared; mod sparse; diff --git a/crates/cellule-ltx/tests/cell/roots/packed.rs b/crates/cellule-ltx/tests/cell/roots/packed.rs index 0a1fb16b..201f4683 100644 --- a/crates/cellule-ltx/tests/cell/roots/packed.rs +++ b/crates/cellule-ltx/tests/cell/roots/packed.rs @@ -28,7 +28,7 @@ async fn packed_root_vector_authenticates_exact_native_bytes_and_every_metadata_ .await .unwrap(); let wire: serde_json::Value = serde_json::from_slice(&root_bytes).unwrap(); - assert_eq!(wire["version"], 2); + assert_eq!(wire["version"], 3); assert_eq!(wire["directory_height"], 0); assert!(wire["directory_inline"].as_str().unwrap().len() <= 4096); assert_eq!(wire["segments"][0]["packed"], true); diff --git a/crates/cellule-ltx/tests/cell/roots/shared.rs b/crates/cellule-ltx/tests/cell/roots/shared.rs new file mode 100644 index 00000000..abc557cd --- /dev/null +++ b/crates/cellule-ltx/tests/cell/roots/shared.rs @@ -0,0 +1,268 @@ +use super::*; + +#[tokio::test] +async fn shared_publication_uploads_once_and_restores_each_exact_cell() { + let directory = tempfile::tempdir().unwrap(); + let store = Store::new(Arc::new(InMemory::new())); + let first = replica(store.clone(), [71; 32], [72; 16]); + let second = replica(store.clone(), [73; 32], [74; 16]); + let mut dbs = Vec::new(); + let mut captures = Vec::new(); + for (i, replica) in [&first, &second].into_iter().enumerate() { + let mut db = Db::open( + &directory.path().join(format!("source-{i}.db")), + Limits::default(), + ) + .unwrap(); + db.transaction(|tx| { + tx.execute_batch(&format!( + "CREATE TABLE items(value INTEGER); INSERT INTO items VALUES({i});" + )) + }) + .unwrap(); + let cuts = db.capture().unwrap(); + captures.push(replica.shared_captures(&cuts).await.unwrap().unwrap()); + dbs.push(db); + } + let appends = CellReplica::upload_shared(captures, directory.path()) + .await + .unwrap(); + assert_eq!(appends.len(), 2); + assert!( + second + .prepare_shared(None, &appends[0], 0, 1) + .await + .is_err() + ); + let mut shared_digest = None; + let mut prepared_roots = Vec::new(); + for (i, (replica, append)) in [&first, &second].into_iter().zip(&appends).enumerate() { + let prepared = replica.prepare_shared(None, append, 0, 1).await.unwrap(); + let objects = replica.reachable_objects(&prepared.root()).await.unwrap(); + let shared = objects + .iter() + .find(|object| object.kind == CellObjectKind::SharedPacked) + .unwrap(); + assert_eq!(*shared_digest.get_or_insert(shared.digest), shared.digest); + assert!(!objects.iter().any(|object| matches!( + object.kind, + CellObjectKind::Ltx | CellObjectKind::Index | CellObjectKind::Packed + ))); + let destination = directory.path().join(format!("restored-{i}.db")); + replica + .open_root(&prepared.root()) + .await + .unwrap() + .restore(&destination) + .await + .unwrap(); + let restored = rusqlite::Connection::open(&destination).unwrap(); + let value: i64 = restored + .query_row("SELECT value FROM items", [], |row| row.get(0)) + .unwrap(); + assert_eq!(value, i as i64); + prepared_roots.push((prepared.root(), destination)); + } + assert_eq!( + first.publication_cost().objects + second.publication_cost().objects, + 3 + ); + for (i, (replica, (root, destination))) in [&first, &second] + .into_iter() + .zip(prepared_roots) + .enumerate() + { + let compacted = replica + .prepare_compaction(&root, 0..1, 9, directory.path()) + .await + .unwrap(); + assert_eq!(compacted.root().position, root.position); + let compacted_destination = directory.path().join(format!("compacted-{i}.db")); + compacted + .verified() + .restore(&compacted_destination) + .await + .unwrap(); + assert_eq!( + std::fs::read(&destination).unwrap(), + std::fs::read(compacted_destination).unwrap() + ); + } + assert_eq!( + first.publication_cost().objects + second.publication_cost().objects, + 7 + ); + let shared = first.scope(); + let layout = CellStorageLayout::new(store.clone(), Path::from("runtime"), [3; 16]); + let path = layout.incarnation_object_path( + &shared.0, + &shared.1, + &shared_digest.unwrap(), + CellObjectKind::SharedPacked, + ); + assert!(path.as_ref().contains("/shared/objects/")); + store.delete(&path).await.unwrap(); + let prepared = second + .prepare_shared(None, &appends[1], 0, 1) + .await + .unwrap(); + assert!(second.reachable_objects(&prepared.root()).await.is_err()); +} + +#[tokio::test] +async fn shared_scope_is_verified_even_when_sibling_native_bytes_are_identical() { + let directory = tempfile::tempdir().unwrap(); + let backend = Arc::new(InMemory::new()); + let store = Store::new(backend.clone()); + let layout = CellStorageLayout::new(store.clone(), Path::from("runtime"), [3; 16]); + let first = replica(store.clone(), [81; 32], [82; 16]); + let second = replica(store.clone(), [83; 32], [84; 16]); + let mut db = Db::open(&directory.path().join("source"), Limits::default()).unwrap(); + db.transaction(|tx| tx.execute_batch("CREATE TABLE t(v); INSERT INTO t VALUES(7)")) + .unwrap(); + let cuts = db.capture().unwrap(); + let appends = CellReplica::upload_shared( + vec![ + first.shared_captures(&cuts).await.unwrap().unwrap(), + second.shared_captures(&cuts).await.unwrap().unwrap(), + ], + directory.path(), + ) + .await + .unwrap(); + let roots = [ + first + .prepare_shared(None, &appends[0], 1, 1) + .await + .unwrap() + .root(), + second + .prepare_shared(None, &appends[1], 1, 1) + .await + .unwrap() + .root(), + ]; + let path = layout.incarnation_object_path( + &roots[1].cell, + &roots[1].incarnation, + &roots[1].digest, + CellObjectKind::Root, + ); + let bytes = backend.get(&path).await.unwrap().bytes().await.unwrap(); + let mut wire: serde_json::Value = serde_json::from_slice(&bytes).unwrap(); + wire["cell"] = serde_json::json!("51".repeat(32)); + wire["incarnation"] = serde_json::json!("52".repeat(16)); + let bytes = Bytes::from(serde_json::to_vec(&wire).unwrap()); + let forged = RootRef { + digest: *blake3::hash(&bytes).as_bytes(), + ..roots[0] + }; + backend + .put( + &layout.incarnation_object_path( + &forged.cell, + &forged.incarnation, + &forged.digest, + CellObjectKind::Root, + ), + bytes.into(), + ) + .await + .unwrap(); + assert!(first.reachable_objects(&forged).await.is_err()); + assert!( + first + .prepare_compaction(&forged, 0..1, 9, directory.path()) + .await + .is_err() + ); + let reopened = first.open_root(&forged).await.unwrap(); + assert!(reopened.paged().read_page(1).await.is_err()); + assert!( + reopened + .restore(&directory.path().join("forged-restore")) + .await + .is_err() + ); + + let object = first + .reachable_objects(&roots[0]) + .await + .unwrap() + .into_iter() + .find(|object| object.kind == CellObjectKind::SharedPacked) + .unwrap(); + let path = layout.incarnation_object_path( + &roots[0].cell, + &roots[0].incarnation, + &object.digest, + object.kind, + ); + let bytes = backend.get(&path).await.unwrap().bytes().await.unwrap(); + for offset in [0, 8, 12, 16, 48, 64, 72, 80, 88, 96, 128, bytes.len() - 1] { + let mut corrupt = bytes.to_vec(); + corrupt[offset] ^= 1; + backend + .put(&path, Bytes::from(corrupt).into()) + .await + .unwrap(); + for root in roots { + assert!( + replica(store.clone(), root.cell, root.incarnation) + .reachable_objects(&root) + .await + .is_err(), + "corruption at {offset}" + ); + } + } + backend + .put(&path, bytes.slice(..bytes.len() - 1).into()) + .await + .unwrap(); + assert!(first.reachable_objects(&roots[0]).await.is_err()); +} + +#[tokio::test] +async fn shared_upload_refuses_cross_store_and_duplicate_ranges() { + let directory = tempfile::tempdir().unwrap(); + let store = Store::new(Arc::new(InMemory::new())); + let first = replica(store.clone(), [91; 32], [92; 16]); + let second = CellReplica::new( + CellStorageLayout::new(store, Path::from("another-prefix"), [3; 16]), + [93; 32], + [94; 16], + Limits::default(), + ) + .unwrap(); + let mut db = Db::open(&directory.path().join("source"), Limits::default()).unwrap(); + db.transaction(|tx| tx.execute_batch("CREATE TABLE t(v); INSERT INTO t VALUES(7)")) + .unwrap(); + let cuts = db.capture().unwrap(); + let inputs = vec![ + first.shared_captures(&cuts).await.unwrap().unwrap(), + second.shared_captures(&cuts).await.unwrap().unwrap(), + ]; + assert!( + CellReplica::upload_shared(inputs, directory.path()) + .await + .is_err() + ); + let inputs = vec![ + first.shared_captures(&cuts).await.unwrap().unwrap(), + first.shared_captures(&cuts).await.unwrap().unwrap(), + ]; + assert!( + CellReplica::upload_shared(inputs, directory.path()) + .await + .is_err() + ); + assert_eq!(first.publication_cost().objects, 0); + assert!(std::fs::read_dir(directory.path()).unwrap().all(|entry| { + !entry + .unwrap() + .file_name() + .to_string_lossy() + .contains("shared-publication") + })); +} diff --git a/crates/cellule-runtime/docs/append-grants.md b/crates/cellule-runtime/docs/append-grants.md new file mode 100644 index 00000000..4bab4719 --- /dev/null +++ b/crates/cellule-runtime/docs/append-grants.md @@ -0,0 +1,72 @@ +# Signed follower append grants + +The receiver can authorize a bounded node-sequence window through fresh +`NodeDirectory` observations, then verify ordinary appends locally. The example +HTTP transport uses this path. `EnrolledPeerVerifier` still represents one +consumed fresh observation; it is never reused as a TTL cache. + +## Signed format + +`CRBGRNT1` has a fixed 376-byte body followed by a 64-byte Ed25519 signature over +`crab.node-append-grant.v1\0` and the BLAKE3 body digest. Integers are unsigned +big-endian; issue/expiry times must fit valid nonnegative milliseconds. No +trailing bytes or alternate encodings are accepted. + +| Body bytes | Binding | +| --- | --- | +| 0–7 | `CRBGRNT1` | +| 8–103 | Fleet, image and release digests, 32 bytes each | +| 104–199 | Leader physical node and boot (16 each), key and certificate digest (32 each) | +| 200–295 | Receiver physical node and boot, key and certificate digest | +| 296–327 | Original physical follower ensemble, two 16-byte slots; unused final slot is zero | +| 328–375 | Epoch, inclusive first/last sequence, coverage floor, issue/expiry time; six u64 values | + +One grant covers at most 512 sequences and five seconds, further bounded by +both observed node advertisement expiries. Issuance reads the leader's exact +mTLS-bound enrollment and the receiver's exact boot record. The sender also +refreshes its original receiver pin when renewing. The receiver's local +monotonic horizon starts before those reads; wall-clock rollback and slow I/O +cannot extend it. This is a bounded authorization horizon, not a device +persistence or clock-skew qualification. + +## Lifecycle and ownership + +1. Install a fresh receiver boot before serving traffic. +2. Acquire the per-lane lifecycle gate before issuance reads or registry changes. +3. Check the current receiver lease, exact original ensemble, epoch and durable + sealed/retired markers. Only this fresh local path can install a registry entry. +4. On append, authenticate signed request bytes and mTLS identity, match its + grant digest, check every frame's sequence and native integrity, and clamp + pruning to the grant's fresh coverage floor. +5. Hold the gate through native `sync_data` and recheck the local grant horizon + and terminal receiver lease before releasing the receipt. +6. Seal and retire use the same gate and existing synced marker/directory + barriers. Marker installation invalidates grants. Uncovered retirement fails; + a lost seal supplies no closure receipt. +7. Collection uses the same gate. A token replay cannot re-register permission + after marker removal: registry creation requires fresh authority inside the + gate, and delayed issuance cannot overtake closure. Restart drops the registry + and uses a new boot, even with a persisted TLS key. + +The registry has 1,024 credits and charges each entry to the existing follower +index budget. Rejected/cancelled issuance evicts empty transient memory slots; +invalid append tokens never create lanes. Dispatched native jobs own their gate +through completion. The example serializes renewal and append per member, +retains signed request/response and pinned mTLS checks, and permits one fresh +renewal plus exact-frame replay after an ambiguous first response. Native +sequence/digest checks reconcile duplicates without another SQL execution. +HTTP deadlines are rechecked after native work. + +## Evidence and limits + +Native tests cover every modified signed byte, key/certificate/session/window +mismatches, fresh-read amortization, expiry, lease loss, original boot rejection, +seal, uncovered retirement, token replay after collection and transient-slot +cleanup. The bounded [write-proof model](../model/README.md) checks lifecycle +interleavings and includes deliberately unsafe fence and expiry configurations. +It abstracts fsync as a successful barrier; physical crash/power-loss behavior, +provider outages and performance gates still require qualification. + +Grants do not select a Cell root, upload a bundle, or grant bucket durability. +The existing per-Cell fenced response and recoverable follower proof contracts +remain authoritative. diff --git a/crates/cellule-runtime/docs/failover-and-followers.md b/crates/cellule-runtime/docs/failover-and-followers.md index f24c958e..73a39739 100644 --- a/crates/cellule-runtime/docs/failover-and-followers.md +++ b/crates/cellule-runtime/docs/failover-and-followers.md @@ -2246,3 +2246,11 @@ Crab must prove its own version because its control model differs: | [`src/follower`](../src/follower) | Follower store, lane records, and quarantine | | [`src/recovery`](../src/recovery) | Recovery claims, manifests, and overlays | | [`src/publication`](../src/publication) | Ordered root publication and object coverage | + +## Bounded append authorization + +The runtime also supports receiver-signed [append grants](append-grants.md). +Fresh issuance binds the original ensemble, both boots and TLS identities; +ordinary appends retain signed-message, local fence, expiry, native integrity +and fsync checks. Seal, retirement and collection serialize with issuance. +A grant is neither a reused enrollment observation nor object durability proof. diff --git a/crates/cellule-runtime/model/README.md b/crates/cellule-runtime/model/README.md index 170d0e37..1655a8d6 100644 --- a/crates/cellule-runtime/model/README.md +++ b/crates/cellule-runtime/model/README.md @@ -56,3 +56,18 @@ their state counts. Negative configurations must identify the configured invariant by name. A passing model run is protocol-assurance evidence only; release and Celld comparison claims still require the qualification receipts described in `../docs/delivery.md`. + +## Shared write proofs + +`check.sh write-proofs` checks `WriteProofs.tla` and three deliberately broken +configurations. The positive model checks immutable-upload versus authoritative +selection, Cell-binding closure/transfer, receiver seal/retirement, original boot +rejection and monotonic grant expiry. Negative runs must report +`BucketCellFence`, `FrozenTail`, and `GrantLifetime` respectively. Lost seal is +stuttering and supplies no receipt. A grant clock unit represents the bounded +local authorization horizon; physical timestamps and signing are abstracted. + +This model constrains the implemented [append grant](../docs/append-grants.md) +lifecycle and the proposed [bundle authority](../../../docs/bundle-coverage-proof.md). +It does not establish a production bundle proof, complete dependency verification, +byte-identical recovery, liveness or performance qualification. diff --git a/crates/cellule-runtime/model/WriteProofs.cfg b/crates/cellule-runtime/model/WriteProofs.cfg new file mode 100644 index 00000000..c483bb45 --- /dev/null +++ b/crates/cellule-runtime/model/WriteProofs.cfg @@ -0,0 +1,4 @@ +CONSTANTS NodeOnlySelector = FALSE IgnoreDurableFence = FALSE WallOnlyExpiry = FALSE +SPECIFICATION Spec +INVARIANTS FrozenTail RetiredCovered GrantLifetime BucketCellFence ClosedEndpointComplete +CHECK_DEADLOCK FALSE diff --git a/crates/cellule-runtime/model/WriteProofs.tla b/crates/cellule-runtime/model/WriteProofs.tla new file mode 100644 index 00000000..6b6f600d --- /dev/null +++ b/crates/cellule-runtime/model/WriteProofs.tla @@ -0,0 +1,100 @@ +------------------------- MODULE WriteProofs ------------------------- +EXTENDS Naturals, Integers, FiniteSets +CONSTANTS NodeOnlySelector, IgnoreDurableFence, WallOnlyExpiry +VARIABLES mono, wall, boot, grantBoot, grantUntil, grant, + accepted, sealed, covered, retired, grantViolation, + cellEpoch, catalogEpoch, selectorVersion, proposedVersion, + proposedCell, uploaded, selected, closedEndpoint, badBucketAck +vars == <> +Init == /\ mono = 0 /\ wall = 0 /\ boot = 1 + /\ grantBoot = 0 /\ grantUntil = 0 /\ grant = FALSE + /\ accepted = 0 /\ sealed = -1 /\ covered = 0 /\ retired = FALSE + /\ grantViolation = FALSE + /\ cellEpoch = 1 /\ catalogEpoch = 1 /\ selectorVersion = 1 + /\ proposedVersion = 0 /\ proposedCell = 0 /\ uploaded = FALSE + /\ selected = 0 /\ closedEndpoint = -1 /\ badBucketAck = FALSE +Issue == /\ ~retired /\ sealed = -1 /\ mono < 3 + /\ grant' = TRUE /\ grantBoot' = boot /\ grantUntil' = mono + 1 + /\ UNCHANGED <> +Append == LET clock == IF WallOnlyExpiry THEN wall ELSE mono + IN /\ grant /\ grantBoot = boot /\ clock < grantUntil /\ accepted < 2 + /\ (IgnoreDurableFence \/ (~retired /\ sealed = -1)) + /\ accepted' = accepted + 1 + /\ grantViolation' = (grantViolation \/ mono >= grantUntil) + /\ UNCHANGED <> +Seal == /\ sealed = -1 /\ sealed' = accepted + /\ UNCHANGED <> +(* Lost seal is stuttering: no receipt or retirement authority is created. *) +Cover == /\ covered < accepted /\ covered' = accepted + /\ UNCHANGED <> +Retire == /\ ~retired /\ covered = accepted + /\ retired' = TRUE + /\ UNCHANGED <> +Restart == /\ boot = 1 /\ boot' = 2 /\ grant' = FALSE + /\ UNCHANGED <> +Tick == /\ mono < 3 /\ mono' = mono + 1 /\ wall' = wall + 1 + /\ UNCHANGED <> +Rollback == /\ wall > 0 /\ wall' = 0 + /\ UNCHANGED <> +Prepare == /\ ~uploaded /\ catalogEpoch = cellEpoch /\ closedEndpoint = -1 + /\ proposedVersion' = selectorVersion /\ proposedCell' = cellEpoch + /\ uploaded' = TRUE + /\ UNCHANGED <> +Select == /\ uploaded /\ selected = 0 + /\ (NodeOnlySelector \/ (proposedVersion = selectorVersion + /\ proposedCell = catalogEpoch /\ closedEndpoint = -1)) + /\ selected' = 1 + /\ badBucketAck' = (badBucketAck \/ proposedCell # cellEpoch) + /\ UNCHANGED <> +CloseBinding == /\ closedEndpoint = -1 + /\ closedEndpoint' = selected + /\ catalogEpoch' = 0 /\ selectorVersion' = selectorVersion + 1 + /\ UNCHANGED <> +Transfer == /\ cellEpoch = 1 /\ closedEndpoint >= 0 + /\ cellEpoch' = 2 + /\ UNCHANGED <> +Next == Issue \/ Append \/ Seal \/ Cover \/ Retire \/ Restart \/ Tick \/ Rollback + \/ Prepare \/ Select \/ CloseBinding \/ Transfer +Spec == Init /\ [][Next]_vars +FrozenTail == sealed = -1 \/ accepted <= sealed +RetiredCovered == ~retired \/ accepted <= covered +GrantLifetime == ~grantViolation +BucketCellFence == ~badBucketAck +ClosedEndpointComplete == closedEndpoint = -1 \/ selected <= closedEndpoint +============================================================================= diff --git a/crates/cellule-runtime/model/WriteProofsNoFence.cfg b/crates/cellule-runtime/model/WriteProofsNoFence.cfg new file mode 100644 index 00000000..5f343b29 --- /dev/null +++ b/crates/cellule-runtime/model/WriteProofsNoFence.cfg @@ -0,0 +1,4 @@ +CONSTANTS NodeOnlySelector = FALSE IgnoreDurableFence = TRUE WallOnlyExpiry = FALSE +SPECIFICATION Spec +INVARIANTS FrozenTail +CHECK_DEADLOCK FALSE diff --git a/crates/cellule-runtime/model/WriteProofsNodeOnly.cfg b/crates/cellule-runtime/model/WriteProofsNodeOnly.cfg new file mode 100644 index 00000000..965be009 --- /dev/null +++ b/crates/cellule-runtime/model/WriteProofsNodeOnly.cfg @@ -0,0 +1,4 @@ +CONSTANTS NodeOnlySelector = TRUE IgnoreDurableFence = FALSE WallOnlyExpiry = FALSE +SPECIFICATION Spec +INVARIANTS BucketCellFence +CHECK_DEADLOCK FALSE diff --git a/crates/cellule-runtime/model/WriteProofsWallOnly.cfg b/crates/cellule-runtime/model/WriteProofsWallOnly.cfg new file mode 100644 index 00000000..2ca33742 --- /dev/null +++ b/crates/cellule-runtime/model/WriteProofsWallOnly.cfg @@ -0,0 +1,4 @@ +CONSTANTS NodeOnlySelector = FALSE IgnoreDurableFence = FALSE WallOnlyExpiry = TRUE +SPECIFICATION Spec +INVARIANTS GrantLifetime +CHECK_DEADLOCK FALSE diff --git a/crates/cellule-runtime/model/check.sh b/crates/cellule-runtime/model/check.sh index 78c00c98..c4b9456b 100755 --- a/crates/cellule-runtime/model/check.sh +++ b/crates/cellule-runtime/model/check.sh @@ -76,6 +76,7 @@ fi run_model() { local config="$1" local expected="$2" + local module="${3:-CellCoordination}" local output local meta_dir="$cache_dir/meta-negative/${config%.cfg}" mkdir -p "$meta_dir" @@ -84,7 +85,7 @@ run_model() { set +e java -cp "$jar" tlc2.TLC -workers 1 -noGenerateSpecTE \ -metadir "$meta_dir" -config "$model_dir/$config" \ - "$model_dir/CellCoordination.tla" | tee "$output" + "$model_dir/$module.tla" | tee "$output" local status=${PIPESTATUS[0]} set -e if [ "$status" -eq 0 ]; then @@ -99,6 +100,14 @@ run_model() { } case "$mode" in + write-proofs) + java -cp "$jar" tlc2.TLC -workers 1 -nowarning -noGenerateSpecTE \ + -metadir "$cache_dir/meta-write-proofs" \ + -config "$model_dir/WriteProofs.cfg" "$model_dir/WriteProofs.tla" + run_model WriteProofsNodeOnly.cfg BucketCellFence WriteProofs + run_model WriteProofsNoFence.cfg FrozenTail WriteProofs + run_model WriteProofsWallOnly.cfg GrantLifetime WriteProofs + ;; fast) java -cp "$jar" tlc2.TLC -workers 1 -depth 6 -nowarning -noGenerateSpecTE \ -metadir "$cache_dir/meta-fast" \ @@ -122,7 +131,7 @@ case "$mode" in run_model CellCoordinationBrokenRelease.cfg RetainedHasOwner ;; *) - echo "usage: $0 {fast|broad|negative}" >&2 + echo "usage: $0 {fast|broad|negative|write-proofs}" >&2 exit 2 ;; esac diff --git a/crates/cellule-runtime/src/cell/actor/acquire.rs b/crates/cellule-runtime/src/cell/actor/acquire.rs index a7aaae26..36f67036 100644 --- a/crates/cellule-runtime/src/cell/actor/acquire.rs +++ b/crates/cellule-runtime/src/cell/actor/acquire.rs @@ -811,6 +811,7 @@ impl CellRuntime { } publisher = publisher.with_node_durability_slot(Arc::clone(&self.inner.node_durability)); publisher = publisher.with_telemetry(self.inner.telemetry.clone()); + publisher = publisher.with_shared_publication(Arc::clone(&self.inner.shared_publication)); self.inner .sender .send(Message::Activate { diff --git a/crates/cellule-runtime/src/cell/actor/requests.rs b/crates/cellule-runtime/src/cell/actor/requests.rs index 346b1ab3..c73e9f74 100644 --- a/crates/cellule-runtime/src/cell/actor/requests.rs +++ b/crates/cellule-runtime/src/cell/actor/requests.rs @@ -702,6 +702,9 @@ pub(super) fn start_admitted_publication( } pub(super) fn is_storage_publication_error(error: &Error) -> bool { + if let Error::Shared(source) = error { + return is_storage_publication_error(source); + } matches!( error, Error::Storage(_) | Error::Ltx(cellule_ltx::LtxError::Storage(_)) diff --git a/crates/cellule-runtime/src/cell/actor/runtime.rs b/crates/cellule-runtime/src/cell/actor/runtime.rs index 045606c6..bd1eee32 100644 --- a/crates/cellule-runtime/src/cell/actor/runtime.rs +++ b/crates/cellule-runtime/src/cell/actor/runtime.rs @@ -155,6 +155,8 @@ impl CellRuntime { let node_lease = Arc::new(node_lease); let unpublished_node_log_bytes = Arc::new(AtomicU64::new(0)); let node_admission = NodeAdmission::default(); + let shared_publication = + crate::publication::SharedPublication::new(resources.clone(), telemetry.clone()); runtime.spawn(run( receiver, pool.clone(), @@ -181,6 +183,7 @@ impl CellRuntime { node_durability: Arc::new(std::sync::RwLock::new(None)), telemetry, unpublished_node_log_bytes, + shared_publication, }), }) } @@ -407,6 +410,7 @@ impl CellRuntime { .map_err(|_| Error::RuntimeClosed)?; let drain = response.await.map_err(|_| Error::RuntimeClosed)?; let workers = self.inner.pool.shutdown().await; + let publication = self.inner.shared_publication.shutdown().await; let durability = match self.node_durability() { Some((_, durability)) => durability.shutdown().await, None => Ok(()), @@ -414,7 +418,11 @@ impl CellRuntime { // Admission and replica work are stopped. Optional fills outlive their // readers, so keep artifacts/executors until accepted fills complete. self.inner.replica_host.drain_cache_fills().await; - receivers.and(drain).and(workers).and(durability) + receivers + .and(drain) + .and(workers) + .and(publication) + .and(durability) } /// Stops new Cell acquisition while existing owners continue serving. diff --git a/crates/cellule-runtime/src/cell/actor/state.rs b/crates/cellule-runtime/src/cell/actor/state.rs index 7c117180..4b7366b2 100644 --- a/crates/cellule-runtime/src/cell/actor/state.rs +++ b/crates/cellule-runtime/src/cell/actor/state.rs @@ -18,6 +18,7 @@ pub(super) struct RuntimeInner { pub(super) node_durability: NodeDurabilitySlot, pub(super) telemetry: crate::fleet::telemetry::CellTelemetryHandle, pub(super) unpublished_node_log_bytes: Arc, + pub(super) shared_publication: Arc, } pub(super) enum RuntimeNodeLease { @@ -449,7 +450,7 @@ pub(super) struct DueResidentCell { /// Preparation admission carries the retry boundary from original dispatch. pub(super) struct PublicationAdmission { - pub(super) replica: cellule_ltx::CellReplica, + pub(super) replica: crate::publication::PublicationPermit, pub(super) fleet_deadline: std::time::Instant, } diff --git a/crates/cellule-runtime/src/fleet/telemetry.rs b/crates/cellule-runtime/src/fleet/telemetry.rs index 444792f5..2a3d52c9 100644 --- a/crates/cellule-runtime/src/fleet/telemetry.rs +++ b/crates/cellule-runtime/src/fleet/telemetry.rs @@ -37,6 +37,23 @@ pub struct PublicationTiming { pub covered_commits: u64, } +/// One bounded shared upload; participating Cells still select roots separately. +#[derive(Clone, Copy, Debug)] +pub struct SharedPublicationTiming { + /// Participating Cell publication inputs. + pub cells: u64, + /// Scoped native LTX/index rows. + pub rows: u64, + /// Exact shared object bytes, including framing. + pub bytes: u64, + /// Oldest input's queue age when the cohort freezes. + pub queue: Duration, + /// File construction, upload and joined scratch cleanup duration. + pub upload: Duration, + /// Whether every scoped uploaded input was produced. + pub succeeded: bool, +} + /// Native follower timing for one dispatched append batch. #[derive(Clone, Copy, Debug, Default)] pub struct FollowerAppendTiming { @@ -223,6 +240,13 @@ pub trait CellTelemetry: Send + Sync { /// record the objects they did upload. fn publication_cost(&self, _objects: u64, _bytes: u64) {} + /// Records shared cohort work without Cell identity labels. + fn shared_publication(&self, _timing: SharedPublicationTiming) {} + + /// Records ordinary-path fallback: `true` means retained/descriptor pressure, + /// `false` means the capture cannot fit the small-object representation. + fn shared_publication_fallback(&self, _pressure: bool) {} + /// Records bytes sent to follower append lanes and whether every lane acknowledged them. fn node_log_append(&self, _acknowledged: bool, _bytes: u64) {} @@ -387,6 +411,18 @@ impl CellTelemetryHandle { } } + pub(crate) fn shared_publication(&self, timing: SharedPublicationTiming) { + if let Some(telemetry) = self.inner.get() { + telemetry.shared_publication(timing); + } + } + + pub(crate) fn shared_publication_fallback(&self, pressure: bool) { + if let Some(telemetry) = self.inner.get() { + telemetry.shared_publication_fallback(pressure); + } + } + pub(crate) fn node_log_append(&self, acknowledged: bool, bytes: u64) { if let Some(telemetry) = self.inner.get() { telemetry.node_log_append(acknowledged, bytes); diff --git a/crates/cellule-runtime/src/follower/grant/mod.rs b/crates/cellule-runtime/src/follower/grant/mod.rs new file mode 100644 index 00000000..f7d08cf7 --- /dev/null +++ b/crates/cellule-runtime/src/follower/grant/mod.rs @@ -0,0 +1,247 @@ +//! Locally registered grants serialize with durable lane closure and collection. +use super::*; +use crate::identity::Digest; +use crate::node::{ + NodeDirectory, + append_grant::{APPEND_GRANT_LIFETIME_MS, NodeAppendGrant}, +}; +use ed25519_dalek::SigningKey; + +#[cfg(test)] +mod tests; + +pub(super) const MAX_APPEND_GRANTS: usize = 1_024; +#[derive(Default)] +pub(super) struct GrantState { + pub(super) current: Option, + pub(super) closed: bool, +} +pub(super) struct RegisteredGrant { + grant: NodeAppendGrant, + expires: Instant, + _memory: IndexReservation, + _slot: tokio::sync::OwnedSemaphorePermit, +} + +// Failed/cancelled authority I/O must not accumulate unauthenticated lane IDs. +// This only evicts an empty in-memory slot. Native files and closed slots stay +// under their ordinary lifecycle; a dispatched job or waiter prevents eviction. +struct TransientLane { + lane: Lane, + slot: LaneState, + lanes: LaneMap, +} +impl Drop for TransientLane { + fn drop(&mut self) { + let Ok(memory) = self.slot.memory.try_lock() else { + return; + }; + let Ok(grant) = self.slot.grant.try_lock() else { + return; + }; + if memory.is_some() || grant.closed || grant.current.is_some() { + return; + } + if let Ok(mut lanes) = self.lanes.lock() + && Arc::strong_count(&self.slot) == 2 + && lanes + .get(&self.lane) + .is_some_and(|slot| Arc::ptr_eq(slot, &self.slot)) + { + lanes.remove(&self.lane); + } + } +} + +/// Identity extracted from the authenticated and signed append request. +#[derive(Clone, Copy)] +pub struct AppendGrantPeer { + /// Original leader boot, independently bound by the request signature. + pub session: SessionId, + /// Certificate digest supplied by the mTLS listener. + pub certificate: Digest, + /// Signing key extracted from the authenticated mTLS certificate. + pub public_key: [u8; 32], +} +/// Fresh authority and receiver lease used for one local grant issuance. +pub struct AppendGrantIssuer<'a> { + /// Canonical fleet/image/release directory. + pub directory: &'a NodeDirectory, + /// Receiver's advertised boot signing key. + pub signing_key: &'a SigningKey, + /// Current local node lease; proof release rechecks its terminal fence. + pub lease: &'a crate::NodeLeaseGuard, + /// Wall-clock sample taken before issuance I/O. + pub now_ms: i64, +} +/// One append carrying an already authenticated grant digest. +pub struct GrantedFollowerAppend { + /// Request sender's authenticated mTLS identity. + pub peer: AppendGrantPeer, + /// Exact original node-log epoch. + pub log_epoch: u64, + /// Digest of the receiver-issued signed grant. + pub grant: Digest, + /// Ordered, checksum-verified native frames; at most 64 per request. + pub frames: Vec, +} + +pub(super) struct GrantAppend { + pub(super) peer: AppendGrantPeer, + pub(super) digest: Digest, + pub(super) lease: crate::NodeLeaseGuard, +} +impl GrantState { + pub(super) fn authorize(&self, request: &GrantAppend, epoch: u64) -> Result { + request.lease.check()?; + let current = self.current.as_ref().ok_or(Error::Fenced)?; + if self.closed + || Instant::now() >= current.expires + || current.grant.digest() != request.digest + { + return Err(Error::Fenced); + } + let grant = ¤t.grant; + grant.authorize( + request.peer.session, + request.peer.certificate, + request.peer.public_key, + epoch, + grant.first_sequence(), + grant.last_sequence(), + )?; + Ok(grant.clone()) + } + pub(super) fn close(&mut self) { + self.closed = true; + self.current = None; + } +} + +impl FollowerStore { + /// Binds grants to this process's fresh boot before the store serves traffic. + /// Restart must use a new session even when its TLS key is persisted. + #[must_use] + pub fn with_append_grant_receiver(mut self, receiver: SessionId) -> Self { + self.grant_receiver = Some(receiver); + self + } + + /// Issues and locally registers one signed window through fresh authority. + /// + /// The lifecycle gate precedes directory I/O and remains held through local + /// registration. Seal, retirement, and collection use that same gate. A + /// replayed signed token cannot install a new registry entry after closure. + pub async fn open_append_grant( + &self, + peer: AppendGrantPeer, + epoch: u64, + first_sequence: u64, + issuer: AppendGrantIssuer<'_>, + ) -> Result { + issuer.lease.check()?; + let receiver = self + .grant_receiver + .ok_or(Error::Node("append grant receiver is not installed"))?; + let lane = Lane { + leader: peer.session, + epoch, + }; + validate_lane(lane)?; + let transient = TransientLane { + lane, + slot: self.lane_lock(lane)?, + lanes: Arc::clone(&self.lanes), + }; + let lock = &transient.slot; + let mut state = lock.grant.clone().lock_owned().await; + if state.closed { + return Err(Error::Fenced); + } + state.current = None; + let slot = self + .grant_slots + .clone() + .try_acquire_owned() + .map_err(|_| Error::Capacity("append grants"))?; + let memory = IndexReservation::new(&self.index_used, 1_024)?; + let started = Instant::now(); + let grant = issuer + .directory + .issue_append_grant( + peer.session, + peer.certificate, + peer.public_key, + receiver, + issuer.signing_key, + first_sequence, + epoch, + issuer.now_ms, + ) + .await?; + let elapsed = i64::try_from(started.elapsed().as_millis()).map_err(|_| Error::Deadline)?; + let now = issuer.now_ms.checked_add(elapsed).ok_or(Error::Deadline)?; + issuer.lease.check()?; + if !grant.valid_at(now) { + return Err(Error::Fenced); + } + let lifetime = grant + .expires_at_ms() + .checked_sub(issuer.now_ms) + .ok_or(Error::Deadline)?; + if !(1..=APPEND_GRANT_LIFETIME_MS).contains(&lifetime) { + return Err(Error::Fenced); + } + let root = self.root.clone(); + let lock = Arc::clone(lock); + let closed = tokio::task::spawn_blocking(move || { + let _native = lock + .lock() + .map_err(|_| Error::Node("follower lane lock poisoned"))?; + let directory = lane_directory(&root, lane); + Ok::( + directory.join("sealed").exists() || directory.join("retired").exists(), + ) + }) + .await + .map_err(Error::FollowerWorkerJoin)??; + if closed { + state.close(); + return Err(Error::Fenced); + } + issuer.lease.check()?; + let expires = started + std::time::Duration::from_millis(lifetime as u64); + if Instant::now() >= expires { + return Err(Error::Fenced); + } + state.current = Some(RegisteredGrant { + grant: grant.clone(), + expires, + _memory: memory, + _slot: slot, + }); + Ok(grant) + } + + /// Appends using its local signed window and the current receiver lease. + /// Every frame remains signed at transport and fsynced at the native barrier. + /// Request watermarks cannot advance beyond the fresh grant coverage floor. + pub async fn append_granted( + &self, + request: GrantedFollowerAppend, + lease: crate::NodeLeaseGuard, + ) -> Result { + self.append_inner( + request.peer.session, + request.log_epoch, + request.frames, + 0, + Some(GrantAppend { + peer: request.peer, + digest: request.grant, + lease, + }), + ) + .await + } +} diff --git a/crates/cellule-runtime/src/follower/grant/tests.rs b/crates/cellule-runtime/src/follower/grant/tests.rs new file mode 100644 index 00000000..e046c24c --- /dev/null +++ b/crates/cellule-runtime/src/follower/grant/tests.rs @@ -0,0 +1,486 @@ +use super::*; +use crate::{ + identity::NodeId, + node::{NODE_LOG_PROTOCOL_VERSION, NodeAdvertisement, NodeCapacity, NodeFailureDomain}, +}; +use cellule_ltx::{CellStorageLayout, Db, Limits, NodeFrameScope, encode_node_frame}; +use cellule_store::{Store, test_support::CountingObjectStore}; +use object_store::{memory::InMemory, path::Path as ObjectPath}; + +const NOW: i64 = 1_000_000; +struct Fixture { + _root: tempfile::TempDir, + follower: FollowerStore, + directory: NodeDirectory, + backend: Arc, + key: SigningKey, + peer: AppendGrantPeer, + receiver: NodeAdvertisement, + lease: crate::NodeLeaseGuard, + db: Db, +} +impl Fixture { + async fn new() -> Self { + let root = tempfile::tempdir().unwrap(); + let key = SigningKey::from_bytes(&[7; 32]); + let backend = Arc::new(CountingObjectStore::new(Arc::new(InMemory::new()))); + let directory = NodeDirectory::new( + CellStorageLayout::new( + Store::new(backend.clone()), + ObjectPath::from("grants"), + [9; 16], + ), + Digest::from_bytes([2; 32]), + Digest::from_bytes([4; 32]), + Digest::from_bytes([5; 32]), + ); + let advert = |session| { + NodeAdvertisement::sign( + NodeId::from_bytes([session; 16]), + SessionId::from_bytes([session; 16]), + "https://node.invalid".into(), + Digest::from_bytes([2; 32]), + Digest::from_bytes([3; 32]), + Digest::from_bytes([4; 32]), + Digest::from_bytes([5; 32]), + &key, + 1, + NOW, + NOW + 10_000, + vec![Digest::from_bytes([6; 32])], + vec![1], + NodeFailureDomain::default(), + NodeCapacity { + free_memory_bytes: 1 << 20, + free_disk_bytes: 1 << 20, + follower_free_bytes: 1 << 20, + follower_retained_bytes: 0, + job_credits: 2, + log_protocol: NODE_LOG_PROTOCOL_VERSION, + }, + ) + .unwrap() + }; + let leader = directory.create(advert(1), NOW).await.unwrap(); + let receiver = advert(2); + directory.create(receiver.clone(), NOW).await.unwrap(); + directory.recruit_log(&leader, 2, 1, 2, NOW).await.unwrap(); + let follower = FollowerStore::open( + root.path().join("followers"), + Limits::default(), + cellule_ltx::DiskBudget::new(8 << 20), + ) + .unwrap() + .with_append_grant_receiver(receiver.session()); + let mut db = Db::open(&root.path().join("source"), Limits::default()).unwrap(); + db.transaction(|tx| tx.execute_batch("CREATE TABLE t(v); INSERT INTO t VALUES(0)")) + .unwrap(); + Self { + _root: root, + follower, + directory, + backend, + key, + peer: AppendGrantPeer { + session: SessionId::from_bytes([1; 16]), + certificate: Digest::from_bytes([3; 32]), + public_key: SigningKey::from_bytes(&[7; 32]).verifying_key().to_bytes(), + }, + receiver, + lease: crate::NodeLeaseGuard::new(NOW, NOW + 10_000).unwrap(), + db, + } + } + async fn grant(&self, first: u64) -> Result { + self.follower + .open_append_grant( + self.peer, + 2, + first, + AppendGrantIssuer { + directory: &self.directory, + signing_key: &self.key, + lease: &self.lease, + now_ms: NOW, + }, + ) + .await + } + fn frame(&mut self, sequence: u64) -> Bytes { + self.db + .transaction(|tx| tx.execute("UPDATE t SET v=v+1", []).map(|_| ())) + .unwrap(); + let cuts = self.db.capture().unwrap(); + let segment = &cuts.segments[0]; + encode_node_frame( + NodeFrameScope { + leader_session: *self.peer.session.as_bytes(), + log_epoch: 2, + node_sequence: sequence, + application: [9; 16], + cell: [4; 32], + incarnation: [5; 16], + cell_epoch: 1, + commit_sequence: sequence, + }, + segment.info().clone(), + Bytes::from(std::fs::read(segment.path()).unwrap()), + Limits::default(), + ) + .unwrap() + .encoded() + .clone() + } + async fn append(&self, grant: &NodeAppendGrant, frames: Vec) -> Result { + self.follower + .append_granted( + GrantedFollowerAppend { + peer: self.peer, + log_epoch: 2, + grant: grant.digest(), + frames, + }, + self.lease.clone(), + ) + .await + } +} + +#[tokio::test] +async fn signed_grant_binds_every_byte_and_amortizes_directory_reads() { + let mut f = Fixture::new().await; + let grant = f.grant(1).await.unwrap(); + let encoded = grant.encode(); + let decoded = NodeAppendGrant::verify(&encoded, &f.receiver, NOW).unwrap(); + assert_eq!(decoded.digest(), grant.digest()); + assert_eq!(grant.last_sequence(), 512); + for index in 0..encoded.len() { + let mut bad = encoded.clone(); + bad[index] ^= 1; + assert!( + NodeAppendGrant::verify(&bad, &f.receiver, NOW).is_err(), + "byte {index}" + ); + } + assert!(NodeAppendGrant::verify(&encoded[..encoded.len() - 1], &f.receiver, NOW).is_err()); + assert!( + NodeAppendGrant::verify(&encoded, &f.receiver, NOW + APPEND_GRANT_LIFETIME_MS).is_err() + ); + assert!( + grant + .authorize( + SessionId::from_bytes([8; 16]), + f.peer.certificate, + f.peer.public_key, + 2, + 1, + 1 + ) + .is_err() + ); + assert!( + grant + .authorize( + f.peer.session, + Digest::from_bytes([8; 32]), + f.peer.public_key, + 2, + 1, + 1 + ) + .is_err() + ); + assert!( + grant + .authorize(f.peer.session, f.peer.certificate, [8; 32], 2, 1, 1) + .is_err() + ); + assert!( + grant + .authorize( + f.peer.session, + f.peer.certificate, + f.peer.public_key, + 3, + 1, + 1 + ) + .is_err() + ); + assert!( + grant + .authorize( + f.peer.session, + f.peer.certificate, + f.peer.public_key, + 2, + 1, + 513 + ) + .is_err() + ); + // The corruption loop consumes real time; renew before checking native I/O. + let grant = f.grant(1).await.unwrap(); + f.backend.reset(); + for sequence in 1..=32 { + let frame = f.frame(sequence); + let receipt = f.append(&grant, vec![frame]).await.unwrap(); + assert_eq!(receipt.durable_through, sequence); + } + assert!(f.backend.requests().is_empty()); + assert!(f.follower.retained_bytes() > 0); +} + +#[tokio::test] +async fn local_seal_and_retirement_exclude_old_grants_and_replayed_issuance() { + let mut f = Fixture::new().await; + let grant = f.grant(1).await.unwrap(); + let first = f.frame(1); + f.append(&grant, vec![first]).await.unwrap(); + // A lost seal did not execute locally. An uncovered retirement must fail + // and preserve the active append witness rather than manufacture closure. + assert!(f.follower.retire(f.peer.session, 2, 0).await.is_err()); + let second = f.frame(2); + f.append(&grant, vec![second.clone()]).await.unwrap(); + assert_eq!( + f.follower + .seal(f.peer.session, 2) + .await + .unwrap() + .durable_through, + 2 + ); + assert!(matches!( + f.append(&grant, vec![second]).await, + Err(Error::Fenced) + )); + f.backend.reset(); + assert!(matches!(f.grant(3).await, Err(Error::Fenced))); + assert!(f.backend.requests().is_empty()); + f.follower.retire(f.peer.session, 2, 2).await.unwrap(); + let retired = f.follower.retired_lanes(i64::MAX, 8).await.unwrap(); + assert_eq!(retired.len(), 1); + assert!( + f.follower + .remove_retired(retired[0], i64::MAX) + .await + .unwrap() + ); + let replay = f.frame(3); + assert!(matches!( + f.append(&grant, vec![replay]).await, + Err(Error::Fenced) + )); +} + +#[tokio::test] +async fn restart_with_persisted_key_rejects_the_original_receiver_grant() { + let mut f = Fixture::new().await; + let grant = f.grant(1).await.unwrap(); + let frame = f.frame(1); + f.append(&grant, vec![frame.clone()]).await.unwrap(); + let restarted = FollowerStore::open( + f._root.path().join("followers"), + Limits::default(), + cellule_ltx::DiskBudget::new(8 << 20), + ) + .unwrap() + .with_append_grant_receiver(SessionId::from_bytes([3; 16])); + assert!(matches!( + restarted + .append_granted( + GrantedFollowerAppend { + peer: f.peer, + log_epoch: 2, + grant: grant.digest(), + frames: vec![frame] + }, + f.lease.clone() + ) + .await, + Err(Error::Fenced) + )); + let receipt = restarted.seal(f.peer.session, 2).await.unwrap(); + assert_eq!(receipt.durable_through, 1); + assert_eq!( + restarted + .read_tail(f.peer.session, 2, 1) + .await + .unwrap() + .len(), + 1 + ); + assert_eq!(restarted.grant_slots.available_permits(), MAX_APPEND_GRANTS); +} + +#[tokio::test] +async fn expiry_and_receiver_lease_loss_cannot_release_a_proof() { + let mut f = Fixture::new().await; + let grant = f.grant(1).await.unwrap(); + let lane = Lane { + leader: f.peer.session, + epoch: 2, + }; + // Deterministically exhaust the local monotonic horizon. Changing a wall + // sample, or replaying the signature, cannot extend this registry deadline. + f.follower + .lane_lock(lane) + .unwrap() + .grant + .lock() + .await + .current + .as_mut() + .unwrap() + .expires = Instant::now(); + let frame = f.frame(1); + assert!(matches!( + f.append(&grant, vec![frame.clone()]).await, + Err(Error::Fenced) + )); + let grant = f.grant(1).await.unwrap(); + f.lease.fence(); + assert!(matches!( + f.append(&grant, vec![frame]).await, + Err(Error::Fenced) + )); + assert!(matches!(f.grant(1).await, Err(Error::Fenced))); +} + +#[tokio::test] +async fn rejected_issuance_does_not_accumulate_empty_lanes_or_index_charges() { + let f = Fixture::new().await; + for byte in 10..40 { + let mut peer = f.peer; + peer.session = SessionId::from_bytes([byte; 16]); + assert!( + f.follower + .open_append_grant( + peer, + 2, + 1, + AppendGrantIssuer { + directory: &f.directory, + signing_key: &f.key, + lease: &f.lease, + now_ms: NOW + } + ) + .await + .is_err() + ); + } + assert!(f.follower.lanes.lock().unwrap().is_empty()); + assert_eq!(*f.follower.index_used.lock().unwrap(), 0); + assert_eq!( + f.follower.grant_slots.available_permits(), + MAX_APPEND_GRANTS + ); +} + +#[tokio::test] +async fn grant_issuance_and_durable_seal_have_one_ordered_endpoint() { + for issue_first in [false, true] { + let mut f = Fixture::new().await; + let original = f.grant(1).await.unwrap(); + let first = f.frame(1); + f.append(&original, vec![first]).await.unwrap(); + let next = f.frame(2); + let lane = f + .follower + .lane_lock(Lane { + leader: f.peer.session, + epoch: 2, + }) + .unwrap(); + let blocked = lane.grant.lock().await; + let mut issue = Box::pin(f.grant(2)); + let mut seal = Box::pin(f.follower.seal(f.peer.session, 2)); + f.backend.reset(); + // Poll each real operation into the FIFO gate, without racing timers. + if issue_first { + assert!(futures_util::poll!(&mut issue).is_pending()); + assert!(futures_util::poll!(&mut seal).is_pending()); + } else { + assert!(futures_util::poll!(&mut seal).is_pending()); + assert!(futures_util::poll!(&mut issue).is_pending()); + } + drop(blocked); + let (issued, sealed) = tokio::join!(issue, seal); + assert_eq!(sealed.unwrap().durable_through, 1); + if issue_first { + let issued = issued.unwrap(); + assert!(matches!( + f.append(&issued, vec![next.clone()]).await, + Err(Error::Fenced) + )); + } else { + assert!(matches!(issued, Err(Error::Fenced))); + assert!(f.backend.requests().is_empty()); + } + assert!(matches!( + f.append(&original, vec![next]).await, + Err(Error::Fenced) + )); + assert_eq!( + f.follower + .read_tail(f.peer.session, 2, 1) + .await + .unwrap() + .len(), + 1 + ); + assert_eq!( + f.follower.grant_slots.available_permits(), + MAX_APPEND_GRANTS + ); + } +} + +#[tokio::test] +async fn only_fresh_grant_coverage_can_release_native_prefix_records() { + let mut f = Fixture::new().await; + let original = f.grant(1).await.unwrap(); + let first = f.frame(1); + let second = f.frame(2); + f.append(&original, vec![first, second.clone()]) + .await + .unwrap(); + let leader = f + .directory + .load(f.peer.session, NOW) + .await + .unwrap() + .unwrap(); + f.directory + .advance_log_coverage(&leader, 1, NOW) + .await + .unwrap(); + // Publication elsewhere cannot mutate a previously issued pruning floor. + assert_eq!( + f.append(&original, vec![second.clone()]) + .await + .unwrap() + .base_sequence, + 1 + ); + let renewed = f.grant(2).await.unwrap(); + assert_eq!(renewed.covered_through(), 1); + f.backend.reset(); + let third = f.frame(3); + let expected = vec![second, third]; + assert_eq!( + f.append(&renewed, expected.clone()) + .await + .unwrap() + .base_sequence, + 2 + ); + f.follower.seal(f.peer.session, 2).await.unwrap(); + assert!(f.follower.read_tail(f.peer.session, 2, 1).await.is_err()); + assert_eq!( + f.follower.read_tail(f.peer.session, 2, 2).await.unwrap(), + expected + ); + assert!(f.backend.requests().is_empty()); +} diff --git a/crates/cellule-runtime/src/follower/mod.rs b/crates/cellule-runtime/src/follower/mod.rs index 5cb9a195..edb1cdd3 100644 --- a/crates/cellule-runtime/src/follower/mod.rs +++ b/crates/cellule-runtime/src/follower/mod.rs @@ -13,6 +13,8 @@ use crate::identity::SessionId; use crate::{Error, Result}; use std::time::Instant; +mod grant; +pub use grant::{AppendGrantIssuer, AppendGrantPeer, GrantedFollowerAppend}; mod directory; mod inventory; mod records; @@ -82,7 +84,16 @@ enum RetirementWatermark { Recovered { active: bool }, } -type LaneState = Arc>>; +struct LaneSlot { + memory: Mutex>, + grant: Arc>, +} +impl LaneSlot { + fn lock(&self) -> std::sync::LockResult>> { + self.memory.lock() + } +} +type LaneState = Arc; type LaneMap = Arc>>; /// Durable contiguous range retained by one follower lane. @@ -154,6 +165,8 @@ pub struct FollowerStore { admission: crate::fleet::admission::NodeAdmission, inventory_scope: [u8; 16], telemetry: CellTelemetryHandle, + grant_receiver: Option, + grant_slots: Arc, } impl FollowerStore { @@ -187,6 +200,8 @@ impl FollowerStore { admission: crate::fleet::admission::NodeAdmission::default(), inventory_scope: rand::random(), telemetry: CellTelemetryHandle::default(), + grant_receiver: None, + grant_slots: Arc::new(tokio::sync::Semaphore::new(grant::MAX_APPEND_GRANTS)), }) } @@ -242,6 +257,18 @@ impl FollowerStore { epoch: u64, frames: Vec, covered_through: u64, + ) -> Result { + self.append_inner(leader, epoch, frames, covered_through, None) + .await + } + + async fn append_inner( + &self, + leader: SessionId, + epoch: u64, + frames: Vec, + covered_through: u64, + grant: Option, ) -> Result { let encoded_bytes = frames .iter() @@ -263,7 +290,19 @@ impl FollowerStore { if !lane_directory(&self.root, lane).exists() { self.admission.check_new_role()?; } - let lock = self.lane_lock(lane)?; + let lock = if grant.is_some() { + // An invalid/replayed token cannot allocate a new lane. Only fresh + // grant issuance installs its locally charged registry entry. + self.lanes + .lock() + .map_err(|_| Error::Node("follower store lock poisoned"))? + .get(&lane) + .cloned() + .ok_or(Error::Fenced)? + } else { + self.lane_lock(lane)? + }; + let grant_state = lock.grant.clone().lock_owned().await; let root = self.root.clone(); let limits = self.limits; let retained = Arc::clone(&self.retained); @@ -278,6 +317,17 @@ impl FollowerStore { let telemetry = self.telemetry.clone(); let frame_count = frames.len() as u64; tokio::task::spawn_blocking(move || { + // Dispatched work owns the lifecycle gate through fsync and proof + // revalidation even when the HTTP or runtime waiter disconnects. + let grant_state = grant_state; + let authorization = grant + .as_ref() + .map(|request| grant_state.authorize(request, epoch)) + .transpose()?; + let covered_through = authorization.as_ref().map_or( + covered_through, + crate::node::append_grant::NodeAppendGrant::covered_through, + ); let mut observation = AppendObservation { started: Instant::now(), telemetry, @@ -308,11 +358,18 @@ impl FollowerStore { &mut state, &scan_counter, &mut observation.timing, + authorization.as_ref(), ) })(); let resize = follower_bytes(&root).and_then(|bytes| retained.resize(bytes).map_err(Error::from)); let result = settle_disk_reservation(result, resize); + let result = result.and_then(|receipt| { + if let Some(request) = &grant { + grant_state.authorize(request, epoch)?; + } + Ok(receipt) + }); if result.is_err() { if let Some(memory) = state.as_mut() { memory.invalidate(); @@ -342,12 +399,14 @@ impl FollowerStore { pub async fn seal(&self, leader: SessionId, epoch: u64) -> Result { let lane = Lane { leader, epoch }; let lock = self.lane_lock(lane)?; + let grant_state = lock.grant.clone().lock_owned().await; let root = self.root.clone(); let limits = self.limits; let retained = Arc::clone(&self.retained); let index_used = Arc::clone(&self.index_used); let scan_counter = clone_scan_counter(&self.scan_counter); tokio::task::spawn_blocking(move || { + let mut grant_state = grant_state; let retained = retained .lock() .map_err(|_| Error::Node("follower disk reservation lock poisoned"))?; @@ -362,6 +421,10 @@ impl FollowerStore { let resize = follower_bytes(&root).and_then(|bytes| retained.resize(bytes).map_err(Error::from)); let result = settle_disk_reservation(result, resize); + let directory = lane_directory(&root, lane); + if directory.join("sealed").exists() || directory.join("retired").exists() { + grant_state.close(); + } if result.is_err() && let Some(memory) = state.as_mut() { @@ -417,11 +480,13 @@ impl FollowerStore { watermark: RetirementWatermark, ) -> Result { let lock = self.lane_lock(lane)?; + let grant_state = lock.grant.clone().lock_owned().await; let root = self.root.clone(); let limits = self.limits; let retained = Arc::clone(&self.retained); let scan_counter = clone_scan_counter(&self.scan_counter); tokio::task::spawn_blocking(move || { + let mut grant_state = grant_state; let retained = retained .lock() .map_err(|_| Error::Node("follower disk reservation lock poisoned"))?; @@ -435,6 +500,10 @@ impl FollowerStore { let resize = follower_bytes(&root).and_then(|bytes| retained.resize(bytes).map_err(Error::from)); let result = settle_disk_reservation(result, resize); + let directory = lane_directory(&root, lane); + if directory.join("sealed").exists() || directory.join("retired").exists() { + grant_state.close(); + } if result.is_ok() { *state = None; } else if let Some(memory) = state.as_mut() { @@ -557,10 +626,12 @@ impl FollowerStore { epoch: candidate.epoch, }; let lock = self.lane_lock(lane)?; + let grant_state = lock.grant.clone().lock_owned().await; let cleanup_lock = Arc::clone(&lock); let root = self.root.clone(); let retained = Arc::clone(&self.retained); let removed = tokio::task::spawn_blocking(move || { + let mut grant_state = grant_state; let retained = retained .lock() .map_err(|_| Error::Node("follower disk reservation lock poisoned"))?; @@ -568,6 +639,9 @@ impl FollowerStore { .lock() .map_err(|_| Error::Node("follower lane lock poisoned"))?; let removed = remove_retired_sync(&root, lane, candidate, retired_before_ms)?; + if removed { + grant_state.close(); + } retained.resize(follower_bytes(&root)?)?; Ok::(removed) }) @@ -589,14 +663,19 @@ impl FollowerStore { Ok(removed) } - fn lane_lock(&self, lane: Lane) -> Result>>> { + fn lane_lock(&self, lane: Lane) -> Result { let mut lanes = self .lanes .lock() .map_err(|_| Error::Node("follower store lock poisoned"))?; Ok(lanes .entry(lane) - .or_insert_with(|| Arc::new(Mutex::new(None))) + .or_insert_with(|| { + Arc::new(LaneSlot { + memory: Mutex::new(None), + grant: Arc::new(tokio::sync::Mutex::new(grant::GrantState::default())), + }) + }) .clone()) } } diff --git a/crates/cellule-runtime/src/follower/records/append.rs b/crates/cellule-runtime/src/follower/records/append.rs index 19d60ec0..e26350ed 100644 --- a/crates/cellule-runtime/src/follower/records/append.rs +++ b/crates/cellule-runtime/src/follower/records/append.rs @@ -17,6 +17,7 @@ pub(in crate::follower) fn append_sync( state: &mut Option, scan_counter: &ScanCounter, timing: &mut FollowerAppendTiming, + grant: Option<&crate::node::append_grant::NodeAppendGrant>, ) -> Result { validate_lane(lane)?; let directory = lane_directory(root, lane); @@ -81,6 +82,13 @@ pub(in crate::follower) fn append_sync( return Err(Error::Node("follower frame changed lane scope")); } let sequence = scope.node_sequence; + if grant.is_some_and(|grant| { + sequence < grant.first_sequence() || sequence > grant.last_sequence() + }) { + return Err(Error::PeerAuthorization( + "follower frame exceeds append grant", + )); + } let digest = frame.digest(); if let Some(existing) = state.records.get(&sequence) { if existing.digest != digest { diff --git a/crates/cellule-runtime/src/node/append_grant/mod.rs b/crates/cellule-runtime/src/node/append_grant/mod.rs new file mode 100644 index 00000000..07ceafb5 --- /dev/null +++ b/crates/cellule-runtime/src/node/append_grant/mod.rs @@ -0,0 +1,264 @@ +//! Receiver-signed append windows. A grant never supplies Cell authority. +use super::*; + +#[cfg(test)] +mod tests; + +const MAGIC: &[u8; 8] = b"CRBGRNT1"; +const DOMAIN: &[u8] = b"crab.node-append-grant.v1\0"; +/// Maximum locally measured lifetime of one append window. +pub const APPEND_GRANT_LIFETIME_MS: i64 = 5_000; +/// Maximum node sequences authorized by one fresh enrollment read. +pub const APPEND_GRANT_SEQUENCES: u64 = 512; +const BODY_BYTES: usize = 8 + 3 * 32 + 2 * (16 + 16 + 32 + 32) + 2 * 16 + 6 * 8; + +/// Immutable receiver signature binding both physical nodes, boots and TLS keys. +/// +/// Only fresh directory authorization can issue a grant. Decoding authenticates +/// a received grant; it cannot install one in the receiver's local registry. +#[derive(Clone)] +pub struct NodeAppendGrant { + body: [u8; BODY_BYTES], + signature: [u8; 64], +} + +impl NodeAppendGrant { + /// Verifies the exact receiver signature and the caller's original boot pin. + pub fn verify(encoded: &[u8], receiver: &NodeAdvertisement, now_ms: i64) -> Result { + if encoded.len() != BODY_BYTES + 64 || &encoded[..8] != MAGIC { + return Err(Error::PeerAuthorization("invalid append grant encoding")); + } + receiver.validate_at(now_ms)?; + let body: [u8; BODY_BYTES] = encoded[..BODY_BYTES] + .try_into() + .map_err(|_| Error::Peer("append grant body"))?; + let signature: [u8; 64] = encoded[BODY_BYTES..] + .try_into() + .map_err(|_| Error::Peer("append grant signature"))?; + receiver + .verifying_key()? + .verify_strict(&signing_bytes(&body), &Signature::from_bytes(&signature)) + .map_err(Error::PeerSignature)?; + let grant = Self { body, signature }; + if grant.scope() != (receiver.fleet, receiver.image, receiver.release) + || grant.receiver() != receiver.session + || grant.receiver_node() != receiver.node + || grant.receiver_certificate() != receiver.certificate + || grant.receiver_key() != receiver.public_key + || grant.expires_at_ms() > receiver.expires_at_ms + || !grant.valid_at(now_ms) + || grant.last_sequence().checked_sub(grant.first_sequence()) + != Some(APPEND_GRANT_SEQUENCES - 1) + || grant.first_sequence() == 0 + || grant.log_epoch() == 0 + || !grant.members().contains(&receiver.node) + { + return Err(Error::PeerAuthorization( + "append grant receiver or window differs", + )); + } + Ok(grant) + } + + /// Canonical fixed-width bytes including the receiver signature. + #[must_use] + pub fn encode(&self) -> Vec { + [self.body.as_slice(), self.signature.as_slice()].concat() + } + /// Digest carried by every signed append request. + #[must_use] + pub fn digest(&self) -> Digest { + Digest::from_bytes(*blake3::hash(&self.encode()).as_bytes()) + } + /// Original leader boot authorized by the grant. + #[must_use] + pub fn leader(&self) -> SessionId { + SessionId::from_bytes(self.bytes(120)) + } + /// Exact receiver boot; a restarted process cannot adopt a registry entry. + #[must_use] + pub fn receiver(&self) -> SessionId { + SessionId::from_bytes(self.bytes(216)) + } + /// Node-log epoch of the authorized original ensemble. + #[must_use] + pub fn log_epoch(&self) -> u64 { + self.number(328) + } + /// Inclusive first sequence of the authorization window. + #[must_use] + pub fn first_sequence(&self) -> u64 { + self.number(336) + } + /// Inclusive final sequence of the authorization window. + #[must_use] + pub fn last_sequence(&self) -> u64 { + self.number(344) + } + /// Fresh authority coverage floor; later request watermarks cannot prune more. + #[must_use] + pub fn covered_through(&self) -> u64 { + self.number(352) + } + /// Signed issue time; local monotonic lifetime starts before issuance I/O. + #[must_use] + pub fn issued_at_ms(&self) -> i64 { + self.number(360) as i64 + } + /// Signed expiry, bounded by both observed node advertisements. + #[must_use] + pub fn expires_at_ms(&self) -> i64 { + self.number(368) as i64 + } + /// Whether wall-clock policy still permits this signed grant. + #[must_use] + pub fn valid_at(&self, now_ms: i64) -> bool { + self.issued_at_ms() >= 0 + && self.issued_at_ms() <= now_ms + && now_ms < self.expires_at_ms() + && self + .expires_at_ms() + .checked_sub(self.issued_at_ms()) + .is_some_and(|ms| (1..=APPEND_GRANT_LIFETIME_MS).contains(&ms)) + } + /// Validates the mTLS identity and frame window on an already signed request. + pub fn authorize( + &self, + leader: SessionId, + certificate: Digest, + key: [u8; 32], + epoch: u64, + first: u64, + last: u64, + ) -> Result<()> { + if leader != self.leader() + || certificate != self.leader_certificate() + || key != self.leader_key() + || epoch != self.log_epoch() + || first > last + || first < self.first_sequence() + || last > self.last_sequence() + { + return Err(Error::PeerAuthorization( + "append grant sender or sequence differs", + )); + } + Ok(()) + } + fn bytes(&self, offset: usize) -> [u8; N] { + let mut out = [0; N]; + out.copy_from_slice(&self.body[offset..offset + N]); + out + } + fn number(&self, offset: usize) -> u64 { + u64::from_be_bytes(self.bytes(offset)) + } + fn scope(&self) -> (Digest, Digest, Digest) { + ( + Digest::from_bytes(self.bytes(8)), + Digest::from_bytes(self.bytes(40)), + Digest::from_bytes(self.bytes(72)), + ) + } + fn receiver_node(&self) -> NodeId { + NodeId::from_bytes(self.bytes(200)) + } + fn receiver_certificate(&self) -> Digest { + Digest::from_bytes(self.bytes(264)) + } + fn receiver_key(&self) -> [u8; 32] { + self.bytes(232) + } + fn leader_certificate(&self) -> Digest { + Digest::from_bytes(self.bytes(168)) + } + fn leader_key(&self) -> [u8; 32] { + self.bytes(136) + } + fn members(&self) -> [NodeId; 2] { + [ + NodeId::from_bytes(self.bytes(296)), + NodeId::from_bytes(self.bytes(312)), + ] + } +} + +fn signing_bytes(body: &[u8]) -> Vec { + [DOMAIN, blake3::hash(body).as_bytes()].concat() +} + +impl NodeDirectory { + #[expect( + clippy::too_many_arguments, + reason = "fresh mTLS and original receiver identities remain explicit" + )] + pub(crate) async fn issue_append_grant( + &self, + leader: SessionId, + certificate: Digest, + key: [u8; 32], + receiver: SessionId, + signing_key: &SigningKey, + first: u64, + epoch: u64, + now_ms: i64, + ) -> Result { + let enrollment = self.peer_verifier(leader, certificate, key, now_ms).await?; + let receiver = self + .load(receiver, now_ms) + .await? + .ok_or(Error::Fenced)? + .advertisement; + let leader = enrollment.into_append_grant_leader(receiver.node, epoch, now_ms)?; + let log = leader.log.as_ref().ok_or(Error::Fenced)?; + if receiver.public_key != signing_key.verifying_key().to_bytes() || first == 0 || now_ms < 0 + { + return Err(Error::PeerAuthorization("append grant issuer differs")); + } + let last = first + .checked_add(APPEND_GRANT_SEQUENCES - 1) + .ok_or(Error::Capacity("append grant sequence"))?; + let expires = now_ms + .checked_add(APPEND_GRANT_LIFETIME_MS) + .ok_or(Error::Deadline)? + .min(leader.expires_at_ms) + .min(receiver.expires_at_ms); + let mut bytes = Vec::with_capacity(BODY_BYTES); + bytes.extend_from_slice(MAGIC); + for digest in [self.fleet, self.image, self.release] { + bytes.extend_from_slice(digest.as_bytes()); + } + for node in [&leader, &receiver] { + bytes.extend_from_slice(node.node.as_bytes()); + bytes.extend_from_slice(node.session.as_bytes()); + bytes.extend_from_slice(&node.public_key); + bytes.extend_from_slice(node.certificate.as_bytes()); + } + let mut members = [[0u8; 16]; 2]; + for (slot, member) in members.iter_mut().zip(log.members()) { + *slot = *member.as_bytes(); + } + for member in members { + bytes.extend_from_slice(&member); + } + for number in [ + epoch, + first, + last, + log.tiered_through(), + now_ms as u64, + expires as u64, + ] { + bytes.extend_from_slice(&number.to_be_bytes()); + } + let body: [u8; BODY_BYTES] = bytes + .try_into() + .map_err(|_| Error::Peer("append grant layout"))?; + let signature = signing_key.sign(&signing_bytes(&body)).to_bytes(); + let grant = NodeAppendGrant { body, signature }; + if !grant.valid_at(now_ms) { + return Err(Error::Fenced); + } + Ok(grant) + } +} diff --git a/crates/cellule-runtime/src/node/append_grant/tests.rs b/crates/cellule-runtime/src/node/append_grant/tests.rs new file mode 100644 index 00000000..6f3f7a0f --- /dev/null +++ b/crates/cellule-runtime/src/node/append_grant/tests.rs @@ -0,0 +1,23 @@ +use super::*; + +#[test] +fn grant_lifetime_rejects_negative_or_wrapped_times() { + let mut grant = NodeAppendGrant { + body: [0; BODY_BYTES], + signature: [0; 64], + }; + let cases = [ + (-1_i64, 1_i64, 0_i64, false), + (0, i64::MIN, 0, false), + (0, 0, 0, false), + (0, APPEND_GRANT_LIFETIME_MS + 1, 0, false), + (10, 20, 9, false), + (10, 20, 20, false), + (10, 20, 10, true), + ]; + for (issued, expires, now, valid) in cases { + grant.body[360..368].copy_from_slice(&(issued as u64).to_be_bytes()); + grant.body[368..376].copy_from_slice(&(expires as u64).to_be_bytes()); + assert_eq!(grant.valid_at(now), valid, "{issued}..{expires} at {now}"); + } +} diff --git a/crates/cellule-runtime/src/node/directory/mod.rs b/crates/cellule-runtime/src/node/directory/mod.rs index 771579df..ffe2e227 100644 --- a/crates/cellule-runtime/src/node/directory/mod.rs +++ b/crates/cellule-runtime/src/node/directory/mod.rs @@ -66,6 +66,17 @@ impl EnrolledPeerVerifier { log::authorize_advertised_append(&self.advertisement, member, log_epoch, covered_through) } + pub(super) fn into_append_grant_leader( + self, + member: NodeId, + epoch: u64, + now_ms: i64, + ) -> Result { + self.advertisement.validate_at(now_ms)?; + log::authorize_advertised_append(&self.advertisement, member, epoch, 0)?; + Ok(self.advertisement) + } + /// Verifies the signed request while rechecking the enrollment's lifetime. pub fn verify( &self, diff --git a/crates/cellule-runtime/src/node/mod.rs b/crates/cellule-runtime/src/node/mod.rs index 246b0177..f72d14e7 100644 --- a/crates/cellule-runtime/src/node/mod.rs +++ b/crates/cellule-runtime/src/node/mod.rs @@ -1,4 +1,5 @@ //! Node advertisements, capacity, the node directory, leases, and the durability log. +pub mod append_grant; pub mod durability; pub mod lease; pub mod log; diff --git a/crates/cellule-runtime/src/publication/mod.rs b/crates/cellule-runtime/src/publication/mod.rs index 1b7d3f30..ad80c213 100644 --- a/crates/cellule-runtime/src/publication/mod.rs +++ b/crates/cellule-runtime/src/publication/mod.rs @@ -11,6 +11,8 @@ use crate::retry::{Backoff, retry_hint, retryable_storage_error}; use crate::{Error, Result}; mod lineage; +mod shared; +pub(crate) use shared::{PublicationPermit, SharedPublication}; const COMPACTION_CHECK_INTERVAL: u8 = 8; const COMPACTION_DEBT_SEGMENTS: usize = 32; @@ -49,6 +51,7 @@ pub struct CellPublisher { node_durability: Option, telemetry: crate::fleet::telemetry::CellTelemetryHandle, lineage_confirmed: Option, + shared_publication: Option>, } enum AppendBase { @@ -88,6 +91,7 @@ impl CellPublisher { node_durability: None, telemetry: crate::fleet::telemetry::CellTelemetryHandle::default(), lineage_confirmed: None, + shared_publication: None, } } @@ -96,6 +100,14 @@ impl CellPublisher { self } + pub(crate) fn with_shared_publication( + mut self, + coordinator: std::sync::Arc, + ) -> Self { + self.shared_publication = Some(coordinator); + self + } + pub(crate) fn with_node_durability_slot(mut self, durability: NodeDurabilitySlot) -> Self { self.node_durability = Some(durability); self @@ -332,14 +344,32 @@ impl CellPublisher { } } - pub(crate) async fn admit_publication(&mut self) -> Result { + pub(crate) async fn admit_publication(&mut self) -> Result { self.check_node_lease()?; let replica = self.replica.clone(); - let admission = replica.admit_root_preparation(); + let shared = self.shared_publication.clone(); + let admission = async { + match shared { + Some(shared) => { + let slot = shared.admit().await?; + // Choose the actor's complete retained range only after + // foreground root capacity becomes available. Release the + // probe before cohort work so no dirty slot waits for its + // uploader or paired compaction admission. + drop(replica.admit_root_preparation().await?); + Ok(PublicationPermit::Shared(slot)) + } + None => replica + .admit_root_preparation() + .await + .map(|replica| PublicationPermit::Direct(Box::new(replica))) + .map_err(Into::into), + } + }; tokio::pin!(admission); loop { tokio::select! { - result = &mut admission => return result.map_err(Into::into), + result = &mut admission => return result, _ = tokio::time::sleep_until(tokio::time::Instant::from_std(self.renew_at)) => { self.renew().await?; } @@ -349,22 +379,115 @@ impl CellPublisher { pub(crate) async fn prepare_admitted_batch( &mut self, - replica: cellule_ltx::CellReplica, + permit: PublicationPermit, cuts: &cellule_ltx::CaptureBatch, commit_sequence: u64, ) -> Result { - let result = self - .prepare_append( - cuts, - commit_sequence, - self.observed.value().schema, - Some(replica), - ) - .await; + let result = match permit { + PublicationPermit::Shared(slot) => { + self.prepare_shared_batch(slot, cuts, commit_sequence).await + } + PublicationPermit::Direct(replica) => { + self.prepare_append( + cuts, + commit_sequence, + self.observed.value().schema, + Some(*replica), + ) + .await + } + }; self.record_publication_cost(); result } + async fn prepare_shared_batch( + &mut self, + slot: tokio::sync::OwnedSemaphorePermit, + cuts: &cellule_ltx::CaptureBatch, + commit_sequence: u64, + ) -> Result { + // Compaction keeps its canonical paired recovery/dirty admission. Do + // not retain a cohort slot while negotiating those scarce permits. + if self + .compaction_pressure(cuts.segments.len()) + .await? + .is_some() + { + drop(slot); + return self + .prepare_append(cuts, commit_sequence, self.observed.value().schema, None) + .await; + } + let coordinator = self + .shared_publication + .clone() + .ok_or(Error::Control("shared publication is unavailable"))?; + let replica = self.replica.clone(); + let submission = coordinator.submit(&replica, cuts, self.scratch_directory.clone(), slot); + tokio::pin!(submission); + let shared = loop { + tokio::select! { + result = &mut submission => break result, + _ = tokio::time::sleep_until(tokio::time::Instant::from_std(self.renew_at)) => self.renew().await?, + } + }; + self.check_node_lease()?; + let shared = match shared { + Ok(shared) => shared, + // A corrupt sibling or one failed shared upload cannot invalidate + // this Cell's independent captures. The canonical factory rechecks + // its own inputs and preserves its own original source on failure. + Err(Error::Shared(source)) if matches!(source.as_ref(), Error::Ltx(_)) => None, + Err(error) => return Err(error), + }; + let Some(shared) = shared else { + return self + .prepare_append(cuts, commit_sequence, self.observed.value().schema, None) + .await; + }; + self.check_node_lease()?; + // Upload neither selects a Cell root nor extends its writer lifetime. + // The final exact Cell CAS revalidates its fence after this cohort wait. + let base = self.observed.value().ltx_root(); + let schema = self.observed.value().schema; + let mut backoff = Backoff::default(); + loop { + let (replica, confirmation) = lineage::replica(self.replica.clone(), &self.authority); + let attempt = + replica.prepare_shared(base.as_ref(), &shared.append, commit_sequence, schema); + tokio::pin!(attempt); + let result = loop { + tokio::select! { + result = &mut attempt => break result, + _ = tokio::time::sleep_until(tokio::time::Instant::from_std(self.renew_at)) => self.renew().await?, + } + }; + match result { + Ok(prepared) => { + self.lineage_confirmed = *confirmation + .lock() + .map_err(|_| Error::Peer("root lineage confirmation lock poisoned"))?; + self.note_append(&prepared); + return Ok(prepared); + } + Err(source) => { + if source.is_cell_graph_limit() { + return self + .prepare_append(cuts, commit_sequence, schema, None) + .await; + } + let error = lineage::error(source); + if retryable_publication_error(&error) { + backoff.wait(runtime_retry_hint(&error)).await; + } else { + return Err(error); + } + } + } + } + } + pub(crate) async fn prepare( &mut self, pending: &crate::cell::executor::PendingCommit, diff --git a/crates/cellule-runtime/src/publication/shared/mod.rs b/crates/cellule-runtime/src/publication/shared/mod.rs new file mode 100644 index 00000000..c51e2716 --- /dev/null +++ b/crates/cellule-runtime/src/publication/shared/mod.rs @@ -0,0 +1,305 @@ +//! One bounded cohort lane. No Cell dirty permit is held while awaiting upload. + +use std::{ + path::PathBuf, + sync::Arc, + time::{Duration, Instant}, +}; + +use cellule_ltx::{ + SHARED_PUBLICATION_BYTES, SHARED_PUBLICATION_ROWS, SharedAppend, SharedCaptures, +}; +use tokio::sync::{OwnedSemaphorePermit, Semaphore, mpsc, oneshot}; + +use crate::{ + Error, Result, + fleet::resource::{ResourceCost, ResourceLedger, ResourceReservation}, +}; + +const COHORT_UPLOADS: usize = 8; +const COHORT_DEADLINE: Duration = Duration::from_millis(1); + +#[cfg(test)] +mod tests; + +pub(crate) enum PublicationPermit { + Direct(Box), + Shared(OwnedSemaphorePermit), +} + +pub(crate) struct SharedPrepared { + pub(crate) append: SharedAppend, + // Index/row tables remain charged through per-Cell root preparation. + _memory: ResourceReservation, +} + +struct Entry { + captures: SharedCaptures, + scratch: PathBuf, + accepted_at: Instant, + memory: ResourceReservation, + _slot: OwnedSemaphorePermit, + reply: oneshot::Sender>, +} + +enum Message { + Capture(Box), + Shutdown, +} + +pub(crate) struct SharedPublication { + sender: mpsc::Sender, + slots: Arc, + resources: ResourceLedger, + telemetry: crate::fleet::telemetry::CellTelemetryHandle, + worker: tokio::sync::Mutex>>>, +} + +impl SharedPublication { + pub(crate) fn new( + resources: ResourceLedger, + telemetry: crate::fleet::telemetry::CellTelemetryHandle, + ) -> Arc { + let (sender, receiver) = mpsc::channel(SHARED_PUBLICATION_ROWS); + Arc::new(Self { + sender, + slots: Arc::new(Semaphore::new(SHARED_PUBLICATION_ROWS)), + resources, + telemetry: telemetry.clone(), + worker: tokio::sync::Mutex::new(Some(tokio::spawn(run(receiver, telemetry)))), + }) + } + + pub(crate) async fn admit(&self) -> Result { + self.slots + .clone() + .acquire_owned() + .await + .map_err(|_| Error::RuntimeClosed) + } + + pub(crate) async fn submit( + &self, + replica: &cellule_ltx::CellReplica, + cuts: &cellule_ltx::CaptureBatch, + scratch: PathBuf, + slot: OwnedSemaphorePermit, + ) -> Result> { + let estimate = cuts.segments.iter().try_fold(16_usize, |total, segment| { + let info = segment.info(); + let row = usize::try_from(info.size_bytes) + .ok() + .and_then(|bytes| { + (info.database_pages as usize) + .checked_mul(60) + .and_then(|index| bytes.checked_add(index)) + }) + .and_then(|bytes| bytes.checked_add(112)) + .ok_or(Error::Capacity("shared publication bytes"))?; + total + .checked_add(row) + .ok_or(Error::Capacity("shared publication bytes")) + })?; + if estimate as u64 > SHARED_PUBLICATION_BYTES + || cuts.segments.len() > SHARED_PUBLICATION_ROWS + { + self.telemetry.shared_publication_fallback(false); + return Ok(None); + } + // Charge before pinning/indexing/coalescing. Multiple compressed cuts + // can expand into a bounded 256 KiB page map before being re-encoded. + let work = if cuts.segments.len() > 1 { + estimate.max(SHARED_PUBLICATION_BYTES as usize) + } else { + estimate + }; + let memory = work + .checked_mul(4) + .and_then(|bytes| bytes.checked_add(132 << 10)) + .ok_or(Error::Capacity("shared publication memory"))?; + let cost = ResourceCost::zero() + .with_retained_bytes(memory) + .with_file_descriptors(cuts.segments.len() + 2); + let memory = match self.resources.try_reserve(cost) { + Ok(memory) => memory, + // Sharing is optional representation reduction. Never wait for + // memory held by roots which need this very lane to finish. + Err(Error::Capacity(_)) => { + self.telemetry.shared_publication_fallback(true); + return Ok(None); + } + Err(error) => return Err(error), + }; + let Some(captures) = replica.shared_captures(cuts).await? else { + self.telemetry.shared_publication_fallback(false); + return Ok(None); + }; + let (reply, response) = oneshot::channel(); + self.sender + .send(Message::Capture(Box::new(Entry { + captures, + scratch, + accepted_at: Instant::now(), + memory, + _slot: slot, + reply, + }))) + .await + .map_err(|_| Error::RuntimeClosed)?; + // The lane, rather than the waiter, owns dispatched storage and scratch. + response.await.map_err(|_| Error::RuntimeClosed)?.map(Some) + } + + pub(crate) async fn shutdown(&self) -> Result<()> { + // Called after actor publication tasks join. Closing earlier could + // strand a committed Cell still waiting to enter its publication lane. + self.slots.close(); + self.sender + .send(Message::Shutdown) + .await + .map_err(|_| Error::RuntimeClosed)?; + if let Some(worker) = self.worker.lock().await.take() { + worker.await.map_err(|source| Error::PeerTransport { + context: "shared publication worker", + source: Box::new(source), + })??; + } + Ok(()) + } +} + +async fn run( + mut receiver: mpsc::Receiver, + telemetry: crate::fleet::telemetry::CellTelemetryHandle, +) -> Result<()> { + let mut pending = None; + let mut uploads = tokio::task::JoinSet::new(); + let mut failure = None; + loop { + while uploads.len() >= COHORT_UPLOADS { + joined_upload(uploads.join_next().await, &mut failure); + } + let first = match pending.take() { + Some(entry) => entry, + None => match receiver.recv().await { + Some(Message::Capture(entry)) => *entry, + Some(Message::Shutdown) | None => break, + }, + }; + let binding = first.captures.storage_binding(); + let started = first.accepted_at; + // Give already-prepared siblings one scheduler turn to enqueue. Never + // add a timer to a lone bucket waiter: provider latency naturally fills + // the bounded queue at load. The real assembly bound is independent of + // an embedding application's paused or advanced Tokio clock. + let deadline = Instant::now() + COHORT_DEADLINE; + tokio::task::yield_now().await; + let mut bytes = 16 + first.captures.encoded_bytes(); + let mut rows = first.captures.rows(); + let mut entries = vec![first]; + let mut shutdown = false; + while rows < SHARED_PUBLICATION_ROWS && bytes < SHARED_PUBLICATION_BYTES { + if Instant::now() >= deadline { + break; + } + let message = match receiver.try_recv() { + Ok(message) => Some(message), + Err(mpsc::error::TryRecvError::Empty) => break, + Err(mpsc::error::TryRecvError::Disconnected) => None, + }; + let entry = match message { + Some(Message::Capture(entry)) => *entry, + Some(Message::Shutdown) | None => { + shutdown = true; + break; + } + }; + if entry.captures.storage_binding() != binding + || bytes + entry.captures.encoded_bytes() > SHARED_PUBLICATION_BYTES + || rows + entry.captures.rows() > SHARED_PUBLICATION_ROWS + { + pending = Some(entry); + break; + } + bytes += entry.captures.encoded_bytes(); + rows += entry.captures.rows(); + entries.push(entry); + } + let age = started.elapsed(); + uploads.spawn(upload_cohort(entries, telemetry.clone(), rows, bytes, age)); + if shutdown { + break; + } + } + while let Some(result) = uploads.join_next().await { + joined_upload(Some(result), &mut failure); + } + failure.map_or(Ok(()), Err) +} + +fn joined_upload( + result: Option>, + failure: &mut Option, +) { + if let Some(Err(source)) = result + && failure.is_none() + { + *failure = Some(Error::PeerTransport { + context: "shared cohort upload task", + source: Box::new(source), + }); + } +} + +async fn upload_cohort( + entries: Vec, + telemetry: crate::fleet::telemetry::CellTelemetryHandle, + rows: usize, + bytes: u64, + age: Duration, +) { + let scratch = entries[0].scratch.clone(); + let mut inputs = Vec::with_capacity(entries.len()); + let mut replies = Vec::with_capacity(entries.len()); + for entry in entries { + inputs.push(entry.captures); + replies.push((entry.reply, entry.memory, entry._slot)); + } + let cells = inputs.len() as u64; + let upload_started = Instant::now(); + let result = cellule_ltx::CellReplica::upload_shared(inputs, &scratch).await; + telemetry.shared_publication(crate::fleet::telemetry::SharedPublicationTiming { + cells, + rows: rows as u64, + bytes, + queue: age, + upload: upload_started.elapsed(), + succeeded: result.is_ok(), + }); + tracing::debug!( + rows, + bytes, + age_us = age.as_micros(), + succeeded = result.is_ok(), + "shared publication cohort" + ); + match result { + Ok(appends) if appends.len() == replies.len() => { + for (append, (reply, memory, _slot)) in appends.into_iter().zip(replies) { + let _ = reply.send(Ok(SharedPrepared { + append, + _memory: memory, + })); + } + } + result => { + let error = Arc::new(match result { + Err(error) => Error::Ltx(error), + Ok(_) => Error::Control("shared publication result count"), + }); + for (reply, _memory, _slot) in replies { + let _ = reply.send(Err(Error::Shared(error.clone()))); + } + } + } +} diff --git a/crates/cellule-runtime/src/publication/shared/tests.rs b/crates/cellule-runtime/src/publication/shared/tests.rs new file mode 100644 index 00000000..6f9aa5e3 --- /dev/null +++ b/crates/cellule-runtime/src/publication/shared/tests.rs @@ -0,0 +1,256 @@ +use super::*; +use cellule_ltx::{CellReplica, CellStorageLayout, Db, Host, Limits}; +use cellule_store::Store; +use object_store::{memory::InMemory, path::Path}; + +fn resources(memory: usize) -> ResourceLedger { + ResourceLedger::new( + ResourceCost::zero() + .with_retained_bytes(memory) + .with_file_descriptors(64), + ) +} + +#[tokio::test] +async fn minimum_host_permits_drain_shared_work_after_a_sibling_waiter_cancels() { + let directory = tempfile::tempdir().unwrap(); + let host = Host::default() + .with_local_disk_budget(cellule_ltx::DiskBudget::new(8 << 20)) + .with_io_slots(Arc::new(Semaphore::new(1))) + .with_job_slots(Arc::new(Semaphore::new(1))) + .with_dirty_slots(Arc::new(Semaphore::new(1))) + .with_recovery_slots(Arc::new(Semaphore::new(1))) + .with_scratch_slots(Arc::new(Semaphore::new(1))); + let disk = host.local_disk_budget(); + let store = Store::new(Arc::new(InMemory::new())); + let ledger = resources(4 << 20); + let coordinator = SharedPublication::new(ledger.clone(), Default::default()); + let mut db = Db::open(&directory.path().join("source"), Limits::default()).unwrap(); + db.transaction(|tx| tx.execute_batch("CREATE TABLE t(v); INSERT INTO t VALUES(7)")) + .unwrap(); + let cuts = db.capture().unwrap(); + let mut replicas = Vec::new(); + let mut responses = Vec::new(); + let mut entries = Vec::new(); + // Preverify both inputs, then enqueue without yielding. The worker sees + // the complete cohort rather than depending on workstation scheduling. + for i in 1..=2 { + let replica = CellReplica::new( + CellStorageLayout::new(store.clone(), Path::from("shared-budget"), [3; 16]), + [i; 32], + [i; 16], + Limits::default(), + ) + .unwrap() + .with_host(host.clone()); + let captures = replica.shared_captures(&cuts).await.unwrap().unwrap(); + let slot = coordinator.admit().await.unwrap(); + let memory = ledger + .try_reserve( + ResourceCost::zero() + .with_retained_bytes(512 << 10) + .with_file_descriptors(3), + ) + .unwrap(); + let (reply, response) = oneshot::channel(); + responses.push(response); + replicas.push(replica); + entries.push(Entry { + captures, + scratch: directory.path().to_owned(), + accepted_at: Instant::now(), + memory, + _slot: slot, + reply, + }); + } + for mut entry in entries { + entry.accepted_at = Instant::now(); + coordinator + .sender + .try_send(Message::Capture(Box::new(entry))) + .unwrap(); + } + drop(responses.pop()); + let response = responses.pop().unwrap(); + let shared = tokio::time::timeout(Duration::from_secs(10), response) + .await + .unwrap() + .unwrap() + .unwrap(); + let prepared = replicas[0] + .prepare_shared(None, &shared.append, 1, 1) + .await + .unwrap(); + assert_eq!( + replicas + .iter() + .map(|replica| replica.publication_cost().objects) + .sum::(), + 2 + ); + assert_eq!(prepared.root().position, cuts.position); + drop(shared); + coordinator.shutdown().await.unwrap(); + assert_eq!(ledger.snapshot().unwrap().used, ResourceCost::zero()); + assert_eq!(disk.used(), 0); + assert!(coordinator.admit().await.is_err()); + assert!(std::fs::read_dir(directory.path()).unwrap().all(|entry| { + !entry + .unwrap() + .file_name() + .to_string_lossy() + .contains("shared-publication") + })); +} + +#[tokio::test] +async fn memory_pressure_falls_back_without_waiting_for_a_dirty_permit() { + let directory = tempfile::tempdir().unwrap(); + let dirty = Arc::new(Semaphore::new(1)); + let host = Host::default().with_dirty_slots(dirty.clone()); + let permit = dirty.clone().acquire_owned().await.unwrap(); + let ledger = resources(1); + let coordinator = SharedPublication::new(ledger.clone(), Default::default()); + let replica = CellReplica::new( + CellStorageLayout::new( + Store::new(Arc::new(InMemory::new())), + Path::from("shared-pressure"), + [3; 16], + ), + [1; 32], + [2; 16], + Limits::default(), + ) + .unwrap() + .with_host(host); + let mut db = Db::open(&directory.path().join("source"), Limits::default()).unwrap(); + db.transaction(|tx| tx.execute_batch("CREATE TABLE t(v); INSERT INTO t VALUES(7)")) + .unwrap(); + let cuts = db.capture().unwrap(); + let slot = coordinator.admit().await.unwrap(); + let result = tokio::time::timeout( + Duration::from_secs(1), + coordinator.submit(&replica, &cuts, directory.path().to_owned(), slot), + ) + .await + .unwrap() + .unwrap(); + assert!(result.is_none()); + assert_eq!( + coordinator.slots.available_permits(), + SHARED_PUBLICATION_ROWS + ); + assert_eq!(ledger.snapshot().unwrap().used, ResourceCost::zero()); + drop(permit); + let prepared = replica.prepare(None, &cuts, 1, 1).await.unwrap(); + assert_eq!(prepared.root().position, cuts.position); + coordinator.shutdown().await.unwrap(); +} + +#[tokio::test] +async fn a_fenced_cell_cannot_select_its_shared_proposal_or_block_a_sibling() { + use crate::{ + control::authority::CellAuthority, + control::{Control, Owner, Transition}, + identity::{CellId, Digest, IncarnationId, SessionId}, + publication::CellPublisher, + }; + let directory = tempfile::tempdir().unwrap(); + let ledger = resources(8 << 20); + let coordinator = SharedPublication::new(ledger.clone(), Default::default()); + let layout = CellStorageLayout::new( + Store::new(Arc::new(InMemory::new())), + Path::from("shared-fencing"), + [3; 16], + ); + let authority = CellAuthority::new(layout.clone()); + let mut db = Db::open(&directory.path().join("source"), Limits::default()).unwrap(); + db.transaction(|tx| tx.execute_batch("CREATE TABLE t(v); INSERT INTO t VALUES(7)")) + .unwrap(); + let cuts = db.capture().unwrap(); + let mut publishers = Vec::new(); + for i in 1..=2 { + let cell = CellId::from_bytes([i; 32]); + let incarnation = IncarnationId::from_bytes([i; 16]); + let control = Control::initial( + cell, + incarnation, + Owner { + session: SessionId::from_bytes([5; 16]), + endpoint: "https://node.invalid".into(), + }, + Digest::from_bytes([4; 32]), + 1, + ) + .unwrap(); + layout + .store() + .create_strict( + &layout.control_path(cell.as_bytes()), + bytes::Bytes::from(control.encode().unwrap()), + ) + .await + .unwrap(); + let observed = authority.load(cell).await.unwrap().unwrap(); + let replica = CellReplica::new( + layout.clone(), + *cell.as_bytes(), + *incarnation.as_bytes(), + Limits::default(), + ) + .unwrap(); + publishers.push( + CellPublisher::new( + replica, + authority.clone(), + observed, + directory.path().to_owned(), + ) + .with_shared_publication(coordinator.clone()), + ); + } + let mut live = publishers.pop().unwrap(); + let mut stale = publishers.pop().unwrap(); + let a = stale.admit_publication().await.unwrap(); + let b = live.admit_publication().await.unwrap(); + let (a, b) = tokio::join!( + stale.prepare_admitted_batch(a, &cuts, 1), + live.prepare_admitted_batch(b, &cuts, 1) + ); + let a = a.unwrap(); + let b = b.unwrap(); + let observed = authority + .load(stale.control().value().cell) + .await + .unwrap() + .unwrap(); + let mut tombstone = observed.value().clone(); + tombstone.state = crate::control::ControlState::Tombstoned; + tombstone.owner = None; + tombstone.epoch += 1; + tombstone.revision += 1; + tombstone.progress += 1; + authority + .transition(&observed, tombstone, Transition::Tombstone) + .await + .unwrap(); + assert!(matches!( + stale.publish_prepared(&a, None).await, + Err(Error::Fenced) + )); + assert_eq!(live.publish_prepared(&b, None).await.unwrap(), b.root()); + assert!( + authority + .load(stale.control().value().cell) + .await + .unwrap() + .unwrap() + .value() + .root + .is_none() + ); + live.replica.reachable_objects(&b.root()).await.unwrap(); + coordinator.shutdown().await.unwrap(); + assert_eq!(ledger.snapshot().unwrap().used, ResourceCost::zero()); +} diff --git a/crates/cellule-runtime/src/recovery/retention/mod.rs b/crates/cellule-runtime/src/recovery/retention/mod.rs index 2c098d50..7d4df1d3 100644 --- a/crates/cellule-runtime/src/recovery/retention/mod.rs +++ b/crates/cellule-runtime/src/recovery/retention/mod.rs @@ -540,6 +540,9 @@ fn immutable_candidate(application_prefix: &Path, location: &Path) -> bool { ["releases", object] | ["catalog", "objects", object] | ["pins", "objects", object] => { json_digest(object) } + ["shared", "objects", object] => object + .strip_suffix(".spack") + .is_some_and(|digest| lower_hex(digest, 64)), ["cells", cell, "inc", incarnation, "objects", object] => { lower_hex(cell, 64) && lower_hex(incarnation, 32) diff --git a/crates/cellule-runtime/src/recovery/retention/tests.rs b/crates/cellule-runtime/src/recovery/retention/tests.rs index 504926af..b4c5e728 100644 --- a/crates/cellule-runtime/src/recovery/retention/tests.rs +++ b/crates/cellule-runtime/src/recovery/retention/tests.rs @@ -606,6 +606,10 @@ fn immutable_path_classifier_accepts_only_version_one_layouts() { &prefix, &layout.catalog_object_path(&[6; 32]), )); + assert!(immutable_candidate( + &prefix, + &layout.incarnation_object_path(&[3; 32], &[4; 16], &[5; 32], CellObjectKind::SharedPacked) + )); assert!(!immutable_candidate( &prefix, &layout.control_path(&[3; 32]), @@ -619,3 +623,146 @@ fn immutable_path_classifier_accepts_only_version_one_layouts() { assert!(GarbageCollectionPolicy::new(0, 1, 0).is_err()); assert!(GarbageCollectionPolicy::new(0, 1, 100_001).is_err()); } + +#[tokio::test] +async fn collection_keeps_a_shared_object_after_its_hot_sibling_compacts() { + let identity = identity(); + let layout = CellStorageLayout::new( + Store::new(Arc::new(InMemory::new())), + Path::from("shared-retention"), + *identity.application().as_bytes(), + ); + let catalog = CellCatalog::new(layout.clone(), identity.tenant()); + let targets = [target(identity, b"hot"), target(identity, b"dormant")]; + let code = Digest::from_bytes([4; 32]); + let directory = tempfile::tempdir().unwrap(); + let mut db = + cellule_ltx::Db::open(&directory.path().join("source"), ReplicaLimits::default()).unwrap(); + db.transaction(|tx| tx.execute_batch("CREATE TABLE t(v); INSERT INTO t VALUES(7)")) + .unwrap(); + let cuts = db.capture().unwrap(); + let mut replicas = Vec::new(); + let mut inputs = Vec::new(); + for (i, target) in targets.iter().enumerate() { + catalog + .provision(CatalogEntry::new(target, CatalogRole::Application, code, 1).unwrap()) + .await + .unwrap(); + let replica = cellule_ltx::CellReplica::new( + layout.clone(), + *target.cell_id().as_bytes(), + [i as u8 + 5; 16], + ReplicaLimits::default(), + ) + .unwrap(); + inputs.push(replica.shared_captures(&cuts).await.unwrap().unwrap()); + replicas.push(replica); + } + let appends = cellule_ltx::CellReplica::upload_shared(inputs, directory.path()) + .await + .unwrap(); + let roots = [ + replicas[0] + .prepare_shared(None, &appends[0], 1, 1) + .await + .unwrap() + .root(), + replicas[1] + .prepare_shared(None, &appends[1], 1, 1) + .await + .unwrap() + .root(), + ]; + let shared = replicas[0] + .reachable_objects(&roots[0]) + .await + .unwrap() + .into_iter() + .find(|object| object.kind == CellObjectKind::SharedPacked) + .unwrap(); + let shared_path = layout.incarnation_object_path( + &roots[0].cell, + &roots[0].incarnation, + &shared.digest, + shared.kind, + ); + let hot_root = replicas[0] + .prepare_compaction(&roots[0], 0..1, 9, directory.path()) + .await + .unwrap() + .root(); + let mut controls = Vec::new(); + for root in [hot_root, roots[1]] { + let control = idle_control( + CellId::from_bytes(root.cell), + IncarnationId::from_bytes(root.incarnation), + root, + code, + ); + layout + .store() + .create_strict( + &layout.control_path(control.cell.as_bytes()), + Bytes::from(control.encode().unwrap()), + ) + .await + .unwrap(); + controls.push(control); + } + let releases = ReleaseStore::new(layout.clone(), identity).unwrap(); + let descriptor = br#"{"runtime":"shared-retention","version":1}"#; + let operation = RequestId::from_bytes([7; 16]); + let prepared = releases + .prepare( + descriptor, + Digest::from_bytes(*blake3::hash(descriptor).as_bytes()), + 0, + &format!("sha256:{}", "a".repeat(64)), + operation, + ) + .await + .unwrap(); + let maintenance = releases + .start_maintenance(prepared.revision(), operation) + .await + .unwrap(); + let collector = CellGarbageCollector::new( + layout.clone(), + identity, + ReplicaLimits::default(), + ReplicaHost::default(), + ) + .unwrap(); + let policy = GarbageCollectionPolicy::new(i64::MAX, 1, 100_000).unwrap(); + collector + .collect(&maintenance, directory.path(), policy) + .await + .unwrap(); + layout.store().head(&shared_path).await.unwrap(); + replicas[1].reachable_objects(&roots[1]).await.unwrap(); + // Only once the complete application reference set drops the dormant root + // can the shared object cross the same maintenance/grace deletion gate. + controls[1].state = ControlState::Tombstoned; + controls[1].epoch += 1; + controls[1].revision += 1; + controls[1].progress += 1; + let authority = CellAuthority::new(layout.clone()); + let observed = authority.load(controls[1].cell).await.unwrap().unwrap(); + authority + .transition( + &observed, + controls[1].clone(), + crate::control::Transition::Tombstone, + ) + .await + .unwrap(); + collector + .collect(&maintenance, directory.path(), policy) + .await + .unwrap(); + assert!(matches!( + layout.store().head(&shared_path).await, + Err(cellule_store::StorageError::NotFound { .. }) + )); + replicas[0].reachable_objects(&hot_root).await.unwrap(); +} diff --git a/docs/bundle-coverage-proof.md b/docs/bundle-coverage-proof.md new file mode 100644 index 00000000..150f0785 --- /dev/null +++ b/docs/bundle-coverage-proof.md @@ -0,0 +1,84 @@ +# Bundle coverage authority decision + +Status: protocol model and implementation design; bundle-based bucket responses +are disabled. Shared payload upload and signed append grants are separate +implemented paths. The [proposal](write-performance-proposal.md) remains the +qualification contract. + +## Cost and latency gap + +The packed implementation measured 1.045 commands per selected Cell root at +2,000 offered bucket writes/s over 1,000 uniform Cells. Root, accumulating +lineage and Cell control still require approximately `3 / 1.045 = 2.87` PUTs +per command before payload and maintenance work. A shared payload changes its +cost to `1 / cohort_commands`; it cannot remove those authority writes. Under +the deterministic uniform schedule, each Cell receives a command every 500 ms. +Waiting for twelve commands per root exceeds the 200-ms bucket latency budget. + +M4 therefore needs an exact recoverable range proof whose selection is shared +across Cells, with asynchronous materialization. It also needs to release proved +capture bodies from retained memory while keeping bounded authenticated locators. +A faster upload without these changes can still accumulate debt and exhaust +admission. + +## Authority and transfer + +Use the existing canonical node record's mutable log authority for both the +binding-catalog digest and selected immutable range head. Lease renewal, +publication, binding closure and node recovery must CAS this same record and +preserve each other's fields. A separate selector checked against a previously +read node advertisement would reinstate a fencing race. + +| Object or capability | Exact meaning | +| --- | --- | +| Cell binding | Cell/incarnation, writer epoch, exact base root/schema/code, permitted node boot/log epoch, and a unique binding identity pinned by Cell control | +| Immutable binding catalog | Complete active bindings plus terminal closure endpoints; closed binding IDs cannot be re-added | +| Immutable range manifest | Contiguous ordered node ranges, exact Cell bindings and commit/transaction intervals, scoped byte extents and digests, complete native outcome/dependency coverage, predecessor head | +| Node selection | Post-upload CAS of both catalog and range head while the original node/log remains Open and every row's binding remains active | +| `BundleCoverageProof` | Opaque capability minted only after exact selection reconciliation and complete dependency verification; uploaded bytes cannot construct it | +| Materialization | Exact logical endpoint reconstructed from base and selected rows; ordinary Cell root/lineage CAS consumes the same proof | +| Closure endpoint | Complete selected prefix frozen in the same CAS that closes the binding; new Cell ownership cannot execute before reconstructing it | + +A live-node Cell transfer must first close its old binding in that node's +canonical record. A proposed upload holding an earlier catalog/head version +then loses its CAS, reloads, and rejects the closed rows. The closed binding +retains all earlier selected rows and its base until exact reconstruction or +retention proof releases them. Only then can Cell authority select the next +writer. Failed-node recovery first fences the whole original node record and +freezes its selected head before closing individual bindings. This ordering +must also govern release, takeover, tombstone, migration, backup and failed +shutdown; a lower-level Cell transition cannot bypass it. + +## Atomic integration required before enabling responses + +| Surface | Required production change and rejection case | +| --- | --- | +| Node advertisement/log codecs | Catalog/head/materialized frontiers preserved by every heartbeat, enrollment, closure and recovery CAS; reject old formats atomically | +| Cell control and acquisition history | Pin exact binding before SQL; close/freeze it before ownership departure; reject live-node late rows | +| Runtime durability gate | Distinct bundle source and proof; exact ticket/row coverage; no proof from upload alone | +| Actor and executor | Advance proven outcome/query endpoint together; retain debt until proof; replace proved capture bodies with bounded immutable locators | +| Materializer | Coalesce selected intervals under host budgets; root and schema stay byte-identical; failure cannot invalidate an earlier bundle ACK | +| Recovery | Verify base, catalog and every required manifest/row; reject omissions, duplicates, conflicting ranges, missing outcomes and origin dependencies | +| Follower retirement | Preserve the distinction between bundle coverage and materialized-root coverage; a sealed record cannot hide an unmaterialized acknowledged suffix | +| Backup/collection | Mark selected ranges, catalog bases, dormant bindings, lineage and pins across all Cells/nodes; require maintenance and grace before deletion | +| Qualification | Owner/receiver death before root materialization, live-node Cell transfer, delayed uploads, late appends, lost CAS responses, missing/corrupt ranges and all-ACK cold retry | + +The manifest needs a bounded file-backed index and a checkpoint strategy for +long histories. Merely chaining every per-command manifest leaves an unbounded +cold-recovery scan. Catalog checkpoints can advance bases only after verified +materialization; complete reference inventory must survive every intermediate +CAS and cancellation. Its object and metadata costs remain in qualification. + +## Protocol evidence + +`WriteProofs.tla` separates upload, exact selection, binding closure and transfer. +The positive configuration enumerated 13,356 distinct states. The node-only +negative configuration produces `Prepare → CloseBinding → Transfer → Select`, +violating `BucketCellFence`; ignoring closure also loses the frozen endpoint. +Separate negative configurations expose receiver append after a durable seal +and grant expiry extended by wall-clock rollback. + +The model assumes complete verified immutable inputs and atomic authority CAS. +It does not prove manifests, Rust adapter ordering, SQLite bytes, cryptographic +identity, provider semantics, fair completion or the full failure matrix. No +bucket response or retention release uses this proposed proof in production. diff --git a/docs/write-performance-delivery.md b/docs/write-performance-delivery.md index ed4e7d3a..5a5d7296 100644 --- a/docs/write-performance-delivery.md +++ b/docs/write-performance-delivery.md @@ -1,8 +1,8 @@ # Write performance implementation and verification -This delivers an implementation slice and a quantified gap report, not the -completed M0–M5 plan. The implementation packs small publication dependencies and provides a pinned -Docker comparison with exact retry and cold-state audits. **Celld write parity +This delivers packed dependencies, shared publication, signed append grants and +a quantified gap report, not the completed M0–M5 plan. A pinned Docker +comparison retains exact retry and cold-state audits. **Celld write parity has not been established.** The [proposal](write-performance-proposal.md) remains the acceptance contract; completing tests or a load run does not pass its gates. @@ -14,7 +14,9 @@ the acceptance contract; completing tests or a load run does not pass its gates. | Small directory leaf lives in the root | Removes its separate PUT, GET and cached-origin HEAD | Canonical leaf validation; 2 KiB leaf and 32 KiB root bounds | | Packed compaction input supplies both scratch streams | One full GET per selected pack instead of full plus range GET | Complete verification; bounded transfer and file ownership through cancellation | | Window telemetry separates response, proof and publication | Logical commands per selected root; capture/checkpoint, worker, peer and sync histograms | Counters and frontiers confer no authority or proof | -| Storage families distinguish owner/receiver enrollment | Summed enrollment GET cost across all three nodes | Each signed peer message still uses fresh authorization | +| Storage families distinguish owner/receiver enrollment | Summed enrollment GET cost across all three nodes | Fresh authorization issues signed windows; every append checks its local window and fence | +| Application-scoped shared capture objects | One payload PUT for up to 64 scoped rows and 256 KiB | Per-Cell root selection, exact recovery and complete reference collection remain mandatory | +| Signed follower append windows | Up to 512 sequences/five seconds per fresh issuance | Pinned mTLS, signed RPCs, fsync, local monotonic expiry and durable closure | | Docker runner and reports preserve failures | Source/binary identities, fresh provider volumes, scheduled-arrival latency, all-ACK audits | Errors, drops, unissued offers, provider failures and failed drain cannot pass | | Arrival producer preserves delayed offers | Final wakeup cannot erase a request scheduled inside the window | Original arrival time still determines latency and completion; full queues count drops | | ACK collection and audit stream bounded records | Disk-backed uniqueness index; 256 queued records and at most 128 GET/retry pairs | Seed, warmup, steady, trailing, overload and recovery successes all reconcile | @@ -25,7 +27,7 @@ compaction composed with an append still needs two packs, the final root, lineage and selection: **five PUTs**. Node coverage and maintenance remain in the window numerator. M1's universal four-PUT gate has therefore not passed. -The current root development format is version 2. The +The current root development format is version 3. The [format specification](../crates/cellule-ltx/docs/packed-root-format.md) covers all readers, producers, sparse-read locators, recovery inventories, backup and collection paths. There is no legacy decoding or automatic migration. @@ -36,11 +38,35 @@ collection paths. There is no legacy decoding or automatic migration. | --- | --- | --- | | M0 | Measurement and comparison harness delivered | Three A/A capacity pairs unverified; storage API totals reconcile, but SDK-internal HTTP retries need provider telemetry | | M1 | Packs, inline leaves and bounded compaction spooling delivered | Ordinary append meets four PUTs; composed compaction needs five. Paired cold/sparse-read guardrail unverified | -| M2 | Existing native grouping and per-Cell coalescing preserved | Shared node publication coordinator not implemented; 0.25 publication PUTs/command not achieved | -| M3 | Fresh enrollment roles, peer phases and follower append measured | Signed append grants and durable grant fences not implemented; 0.05 enrollment GETs/command not achieved | -| M4 | Existing authority-pinned Cell roots remain the object proof | Bundle coverage proof, transfer and collection protocol not implemented | +| M2 | Bounded file-backed shared publication coordinator implemented; exact scope, restore, cancellation, minimum-budget and dormant-sibling retention checks added | Full isolated contributor checks pass; paired measurement pending. Three per-Cell authority PUTs remain | +| M3 | Signed 512-sequence/five-second grants, bounded local registry, lifecycle gate and signed HTTP fixture implemented | Full isolated contributor checks, native race/pruning checks and wire tests pass; 0.05 enrollment GETs/command measurement pending | +| M4 | Binding/selector model and [authority decision](bundle-coverage-proof.md) delivered; unsafe node-only selection has a required counterexample | Production bundle proof, atomic transfer/recovery/collection and bundle ACKs not implemented | | M5 | Matched Fleet/read points and target stress with exact ACK audits delivered | Three repetitions, read/failure/overload matrix and absolute/relative parity unverified | +## Shared publication checkpoint + +Shared publication is implemented in the runtime and LTX layer, with the +existing fenced per-Cell response gate. It uses one 64-entry lane, a 1-ms assembly +bound with immediate idle flush and a 256-KiB shared object bound. Fixed cohort counters and cumulative +queue/upload histograms permit windowed comparison. The new format requires a +fresh isolated prefix and coordinated deployment of all producers and consumers. + +The prior evidence below measures the packed implementation, **not this shared +coordinator**. Its improvement percentages must not be attributed to M2. New +source identities, checks and performance results will be recorded separately. +Signed append grants are implemented with a fresh issuance path and local +durable closure gates. Bundle coverage ACKs remain disabled: the proposed +Cell-binding catalog, atomic transfer, exact range recovery and collection +contracts still need production integration. + +The final isolated snapshot passed 1,857 workspace tests (38 documented tests +ignored), 58 local LTX tests without replica features, all-target/all-feature +checking, Rust 1.99 Clippy with warnings denied, API documentation and all +boundary/layout/document/SQL-peer/Python gates. The write-proof model checked +13,356 distinct states and all three required unsafe counterexamples. These +checks validate the implementation contracts; they do not qualify throughput, +physical-device durability or bundle-based bucket acknowledgments. + ## Evidence The fixture is **SQL application parity**, not the user's bounded KV workload: @@ -387,18 +413,20 @@ multipart calls. SDK-internal retries need provider telemetry for exact HTTP cou recovery, backup and collection worker together. 3. Retained version-1 data needs a separately verified logical export/rebuild using the old binary. No automatic conversion route is delivered here. -4. A rollback binary cannot read version-2 roots. Use a verified logical +4. A rollback binary cannot read version-3 roots. Use a verified logical export/rebuild if available; otherwise preserve artifacts and roll forward. Bucket listing and completed uploads never select authority. ## Remaining delivery Complete M0's reproducibility gate before attributing sustainable-rate changes -to the framework. Then implement M2's bounded node publication coordinator, -complete cross-Cell inventory and per-Cell reconciliation. Measure the selection -floor before deciding M4. M3 needs scoped signed grants and receiver fences with -seal, retire, restart, expiry and lease-loss tests; a TTL cache of consumed peer -verifiers does not implement that contract. +to the framework. Measure M2 cohort fill, per-Cell selection cost and debt, and +M3's summed issuance GETs and renewal stalls. The shared coordinator and signed +grant protocol are implemented; their exit budgets require measured evidence. +The authority cost floor requires M4's atomic binding/selector, transfer, +recovery and collection integration before shared coverage can support bucket +acknowledgments. The [authority decision](bundle-coverage-proof.md) defines that +integration; it is not an enabled response path. Finally run three paired five-minute repetitions, bounded KV, read-only and mixed/hot-read guardrails, 1.5× overload with immediate recovery, owner loss diff --git a/scripts/perf/README.md b/scripts/perf/README.md index 3ce4db89..f69de848 100644 --- a/scripts/perf/README.md +++ b/scripts/perf/README.md @@ -103,7 +103,10 @@ indices and fixed bound, preserving 100-us resolution and overflow. The client primes exporters and sampling connections before warmup. Use `--telemetry off` only to measure exporter overhead; that diagnostic cannot qualify. Compare response, confirmation, fleet-proof, capture/checkpoint, and publication timings -separately instead of attributing publication time to the command ACK. +separately instead of attributing publication time to the command ACK. Shared +publication exports window cohort/cell/row/byte and fallback counters plus +`shared_queue` and `shared_upload` histograms. Cohort fill alone does not prove +a lower authority cost or a sustainable capacity increase. The producer emits every offer scheduled inside the window even if its final wakeup is late. It preserves the original due time: lateness remains in the diff --git a/scripts/perf/build.py b/scripts/perf/build.py index cdd80b5d..9625c05d 100644 --- a/scripts/perf/build.py +++ b/scripts/perf/build.py @@ -118,7 +118,14 @@ def build(args): ] for name in names: (source / name).parent.mkdir(parents=True, exist_ok=True) - shutil.copy2(ROOT / name, source / name) + if (name.startswith('crates/cellule-runtime/') and name != 'crates/cellule-runtime/src/fleet/telemetry.rs') or name.startswith('crates/cellule-axum/examples/fleet/'): + # Freeze the audited actor/log measurement hooks. Copying the + # candidate actor now would silently install shared publication + # into the baseline rather than measure main's implementation. + (source / name).write_bytes(subprocess.check_output( + ['git', 'show', 'a9e743ea148eb4e76b30f2b0c62f42dbce04f8bd:' + name], cwd=ROOT)) + else: + shutil.copy2(ROOT / name, source / name) overlay[name] = sha(source / name) manifest = {str(path.relative_to(source)): sha(path) for path in source.rglob('*') if path.is_file()} (destination / 'framework-source.json').write_text(json.dumps(manifest, sort_keys=True, indent=2) + '\n') diff --git a/scripts/perf/report.py b/scripts/perf/report.py index 703a45b7..0f01cfaf 100644 --- a/scripts/perf/report.py +++ b/scripts/perf/report.py @@ -136,7 +136,8 @@ def metric_delta(directory): for outcome, count in values.get('outcomes', {}).items(): if sum((family[operation]['outcomes'][outcome] for family in families.values())) != count: raise ValueError(f'unclassified {operation}.{outcome}') - output.append({'url': a['url'], 'sample_elapsed_ns': subtract(first['sample_elapsed_ns'], last['sample_elapsed_ns']), 'start_request_ms': [a['request_started_ms'], a['request_finished_ms']], 'end_request_ms': [b['request_started_ms'], b['request_finished_ms']], 'storage_families': families, 'storage': totals, 'histograms': {name: histogram_delta(first['histograms'][name], value) for name, value in last['histograms'].items()}, 'publication': subtract({name: first['writes'][name] for name in ('selected_roots', 'materialized_commits', 'publication_failures', 'uploaded_objects', 'uploaded_bytes')}, {name: last['writes'][name] for name in ('selected_roots', 'materialized_commits', 'publication_failures', 'uploaded_objects', 'uploaded_bytes')}), 'runtime_start': first.get('runtime'), 'runtime_end': last.get('runtime')}) + shared = subtract(first.get('shared_publication', {}), last.get('shared_publication', {})) + output.append({'shared_publication': shared, 'url': a['url'], 'sample_elapsed_ns': subtract(first['sample_elapsed_ns'], last['sample_elapsed_ns']), 'start_request_ms': [a['request_started_ms'], a['request_finished_ms']], 'end_request_ms': [b['request_started_ms'], b['request_finished_ms']], 'storage_families': families, 'storage': totals, 'histograms': {name: histogram_delta(first['histograms'][name], value) for name, value in last['histograms'].items()}, 'publication': subtract({name: first['writes'][name] for name in ('selected_roots', 'materialized_commits', 'publication_failures', 'uploaded_objects', 'uploaded_bytes')}, {name: last['writes'][name] for name in ('selected_roots', 'materialized_commits', 'publication_failures', 'uploaded_objects', 'uploaded_bytes')}), 'runtime_start': first.get('runtime'), 'runtime_end': last.get('runtime')}) return {'available': True, 'endpoints': output} def cost_report(metrics, successes): @@ -148,7 +149,11 @@ def cost_report(metrics, successes): 'multipart_start', 'multipart_part', 'multipart_complete', 'multipart_abort')} families = {} roots = commits = 0 + shared = {} for endpoint in endpoints: + for name, value in endpoint.get('shared_publication', {}).items(): + if isinstance(value, (int, float)): + shared[name] = shared.get(name, 0) + value publication = endpoint['publication'] roots += publication['selected_roots'] commits += publication['materialized_commits'] @@ -172,6 +177,9 @@ def cost_report(metrics, successes): 'get_attempts_including_ranges_per_command': (operations['get']['attempts'] + operations['range']['attempts']) / successes, 'mutation_request_successes_per_command': sum(operations[name]['successes'] for name in ('put', 'copy', 'multipart_start', 'multipart_part', 'multipart_complete', 'multipart_abort')) / successes, 'observation_scope': 'storage API calls; provider SDK internal retries are not individually instrumented', + 'shared_publication': shared, + 'shared_cells_per_cohort': shared.get('cells', 0) / shared['cohorts'] if shared.get('cohorts') else None, + 'shared_bytes_per_cohort': shared.get('bytes', 0) / shared['cohorts'] if shared.get('cohorts') else None, 'selected_roots': roots, 'materialized_commits': commits, 'materialized_commits_per_selected_root': commits / roots if roots else None, 'trailing_publication_included': False, diff --git a/scripts/perf/run.py b/scripts/perf/run.py index f71e9cb2..43b2a0f3 100644 --- a/scripts/perf/run.py +++ b/scripts/perf/run.py @@ -105,11 +105,14 @@ def get(path, query=''): if root['version'] != expected or root['cell'] != cell or root['incarnation'] != control['incarnation']: raise SystemExit('persisted root codec or scope differs from source and control') packed = [segment for segment in root['segments'] if segment.get('packed')] -if expected == 2: +if expected in (2, 3): if not packed: raise SystemExit('small-root fixture did not persist a packed dependency') - body = get('/comparison/' + objects + packed[0]['object_digest'] + '.pack') - if body[:8] != b'CRBPACK1' or len(body) > 256 * 1024: + segment = packed[0] + shared = segment.get('shared', False) + prefix = sys.argv[1] + '/cells/v1/apps/' + '03' * 16 + '/shared/objects/' if shared else objects + body = get('/comparison/' + prefix + segment['object_digest'] + ('.spack' if shared else '.pack')) + if body[:8] != (b'CRBSH001' if shared else b'CRBPACK1') or len(body) > 256 * 1024: raise SystemExit('packed dependency has an invalid header or bound') print(json.dumps({'root_path': root_path, 'root_sha256': hashlib.sha256(raw).hexdigest(), 'version': root['version'], 'packed_segments': len(packed), 'inline_directory': root.get('directory_inline') is not None, 'pass': True})) ''' From 4778f2728fb13b9545642d7768be9490fa5ace9f Mon Sep 17 00:00:00 2001 From: forhappy Date: Wed, 7 Oct 2026 02:48:20 -0700 Subject: [PATCH 002/102] Fix shared publication admission and queued-frame append grants --- .github/workflows/coordination-model.yml | 5 ++ crates/cellule-ltx/docs/packed-root-format.md | 5 ++ crates/cellule-ltx/src/replica/shared/mod.rs | 6 +- crates/cellule-runtime/docs/append-grants.md | 5 ++ crates/cellule-runtime/src/cell/actor/mod.rs | 5 +- crates/cellule-runtime/src/cell/worker/mod.rs | 3 + crates/cellule-runtime/src/fleet/resource.rs | 68 ++++++++++++++++++- .../src/follower/grant/tests.rs | 29 ++++++++ .../src/node/append_grant/mod.rs | 6 +- .../src/publication/shared/mod.rs | 2 +- .../src/publication/shared/tests.rs | 59 +++++++++++++++- .../runtime/lifecycle/durability/admission.rs | 2 + .../tests/runtime/lifecycle/residency.rs | 1 + docs/write-performance-delivery.md | 30 +++++++- scripts/perf/report.py | 33 ++++++++- scripts/tests/test_perf_report.py | 48 +++++++++++++ 16 files changed, 295 insertions(+), 12 deletions(-) diff --git a/.github/workflows/coordination-model.yml b/.github/workflows/coordination-model.yml index 4521d2b7..f3515615 100644 --- a/.github/workflows/coordination-model.yml +++ b/.github/workflows/coordination-model.yml @@ -6,6 +6,10 @@ on: - "crates/cellule-runtime/model/**" - "crates/cellule-runtime/src/coordination/mod.rs" - "crates/cellule-runtime/src/coordination/**" + - "crates/cellule-runtime/src/control/**" + - "crates/cellule-runtime/src/follower/**" + - "crates/cellule-runtime/src/node/**" + - "crates/cellule-runtime/src/publication/**" - ".github/workflows/coordination-model.yml" schedule: - cron: "17 3 * * 1" @@ -32,6 +36,7 @@ jobs: chmod +x crates/cellule-runtime/model/check.sh crates/cellule-runtime/model/check.sh fast crates/cellule-runtime/model/check.sh negative + crates/cellule-runtime/model/check.sh write-proofs broad: if: github.event_name != 'pull_request' diff --git a/crates/cellule-ltx/docs/packed-root-format.md b/crates/cellule-ltx/docs/packed-root-format.md index efd6b3c5..f2db57a7 100644 --- a/crates/cellule-ltx/docs/packed-root-format.md +++ b/crates/cellule-ltx/docs/packed-root-format.md @@ -66,6 +66,11 @@ must have the same provider identity and application prefix. Large cuts, compaction, or insufficient retained-memory admission use ordinary preparation. Cancelled waiters leave dispatched upload and scratch cleanup owned by the lane. The lane joins after all Cell actors and publishers during shutdown. +The worker pool reserves 4,224 node publication descriptors in addition to the +eight handles per resident Cell. Only publication pins can consume this headroom; +Cell and reader admission retains its ordinary descriptor bound. Aggregate +descriptor observations include both classes. Retained RAM and disk limits +remain separate. Upload grants no durability authority. Each Cell independently prepares its root, accumulates lineage and performs its existing fenced control CAS. Failed diff --git a/crates/cellule-ltx/src/replica/shared/mod.rs b/crates/cellule-ltx/src/replica/shared/mod.rs index ba8a6a8b..22f1b90a 100644 --- a/crates/cellule-ltx/src/replica/shared/mod.rs +++ b/crates/cellule-ltx/src/replica/shared/mod.rs @@ -280,7 +280,11 @@ impl CellReplica { if offset != size || file.file_len()? != size { return Err(LtxError::LTXCorrupted); } - file.sync_all()?; + // This file is only a transient upload source, never restart + // state or a durability proof. Close the writer before origin + // reads; put_source still verifies its exact length and digest. + // Native captures and follower logs retain their own barriers. + drop(file); let digest = *hasher.finalize().as_bytes(); for append in &mut appends { for segment in &mut append.segments { diff --git a/crates/cellule-runtime/docs/append-grants.md b/crates/cellule-runtime/docs/append-grants.md index 4bab4719..d02b409a 100644 --- a/crates/cellule-runtime/docs/append-grants.md +++ b/crates/cellule-runtime/docs/append-grants.md @@ -29,6 +29,11 @@ monotonic horizon starts before those reads; wall-clock rollback and slow I/O cannot extend it. This is a bounded authorization horizon, not a device persistence or clock-skew qualification. +The signed pruning floor is the lower of fresh authoritative object coverage +and the sequence immediately before the requested window. Publication can +advance past frames already queued by the source; retaining their first +witness preserves the source's exact receipt contract through that race. + ## Lifecycle and ownership 1. Install a fresh receiver boot before serving traffic. diff --git a/crates/cellule-runtime/src/cell/actor/mod.rs b/crates/cellule-runtime/src/cell/actor/mod.rs index 80ece725..ddfa53b3 100644 --- a/crates/cellule-runtime/src/cell/actor/mod.rs +++ b/crates/cellule-runtime/src/cell/actor/mod.rs @@ -215,13 +215,14 @@ impl CellRuntimeStats { self.resident_capacity_bytes } - /// Returns file descriptors reserved by active Cells in the shared ledger. + /// Returns reserved Cell, reader and node publication descriptors. #[must_use] pub const fn file_descriptors(self) -> usize { self.file_descriptors } - /// Returns the active-Cell file-descriptor ceiling in the shared ledger. + /// Returns the combined ordinary-handle and dedicated publication ceilings. + /// Publication headroom cannot admit more Cell or reader handles. #[must_use] pub const fn file_descriptor_capacity(self) -> usize { self.file_descriptor_capacity diff --git a/crates/cellule-runtime/src/cell/worker/mod.rs b/crates/cellule-runtime/src/cell/worker/mod.rs index 0bd874f0..132c04be 100644 --- a/crates/cellule-runtime/src/cell/worker/mod.rs +++ b/crates/cellule-runtime/src/cell/worker/mod.rs @@ -171,6 +171,9 @@ impl SqlWorkerPool { .with_file_descriptors( max_active_cells.saturating_mul(ACTIVE_CELL_FILE_DESCRIPTORS), ) + .with_publication_file_descriptors( + crate::fleet::resource::PUBLICATION_FILE_DESCRIPTORS, + ) .with_worker_jobs(worker_count) .with_primitive_jobs(worker_count) .with_hydration_jobs(HYDRATION_JOB_CAPACITY), diff --git a/crates/cellule-runtime/src/fleet/resource.rs b/crates/cellule-runtime/src/fleet/resource.rs index b8bc05cd..5372f339 100644 --- a/crates/cellule-runtime/src/fleet/resource.rs +++ b/crates/cellule-runtime/src/fleet/resource.rs @@ -8,6 +8,12 @@ pub(crate) const ACTIVE_CELL_NATIVE_BYTES: usize = 64 * 1024 as usize; /// Persistent database, WAL, SHM and capture descriptors reserved per active Cell. pub const ACTIVE_CELL_FILE_DESCRIPTORS: usize = 8; +/// Dedicated node publication pins in addition to resident SQLite/capture handles. +/// One bounded lane's worst-case input/temporary descriptors share a finite +/// ceiling across runtimes on the same pool. Cell and reader admission cannot +/// consume it; both classes still appear in aggregate descriptor observations. +pub const PUBLICATION_FILE_DESCRIPTORS: usize = + cellule_ltx::SHARED_PUBLICATION_ROWS * (cellule_ltx::SHARED_PUBLICATION_ROWS + 2); // Conservatively cover the entire 8 MiB shared page cache plus 4 MiB for // SQLite, fetch/decode buffers and view metadata. Do not assume another view // or writer pays for the cache; measured sharing may reduce this charge later. @@ -21,6 +27,7 @@ pub struct ResourceCost { active_cells: usize, resident_bytes: usize, file_descriptors: usize, + publication_file_descriptors: usize, retained_bytes: usize, disk_bytes: u64, worker_jobs: usize, @@ -52,6 +59,7 @@ impl ResourceCost { active_cells: 0, resident_bytes: 0, file_descriptors: 0, + publication_file_descriptors: 0, retained_bytes: 0, disk_bytes: 0, worker_jobs: 0, @@ -77,10 +85,11 @@ impl ResourceCost { self.resident_bytes } - /// Returns the open file descriptors. + /// Returns ordinary handles plus the separately bounded publication pins. #[must_use] pub const fn file_descriptors(self) -> usize { self.file_descriptors + .saturating_add(self.publication_file_descriptors) } /// Returns the bytes retained beyond the active set. @@ -157,13 +166,18 @@ impl ResourceCost { self } - /// Sets the open file descriptors. + /// Sets ordinary Cell/reader handles; node publication has its own bound. #[must_use] pub const fn with_file_descriptors(mut self, descriptors: usize) -> Self { self.file_descriptors = descriptors; self } + pub(crate) const fn with_publication_file_descriptors(mut self, descriptors: usize) -> Self { + self.publication_file_descriptors = descriptors; + self + } + /// Sets the Cell count. #[must_use] pub const fn with_active_cells(mut self, cells: usize) -> Self { @@ -239,6 +253,9 @@ impl ResourceCost { active_cells: self.active_cells.checked_add(other.active_cells)?, resident_bytes: self.resident_bytes.checked_add(other.resident_bytes)?, file_descriptors: self.file_descriptors.checked_add(other.file_descriptors)?, + publication_file_descriptors: self + .publication_file_descriptors + .checked_add(other.publication_file_descriptors)?, retained_bytes: self.retained_bytes.checked_add(other.retained_bytes)?, disk_bytes: self.disk_bytes.checked_add(other.disk_bytes)?, worker_jobs: self.worker_jobs.checked_add(other.worker_jobs)?, @@ -257,6 +274,9 @@ impl ResourceCost { active_cells: self.active_cells.checked_sub(other.active_cells)?, resident_bytes: self.resident_bytes.checked_sub(other.resident_bytes)?, file_descriptors: self.file_descriptors.checked_sub(other.file_descriptors)?, + publication_file_descriptors: self + .publication_file_descriptors + .checked_sub(other.publication_file_descriptors)?, retained_bytes: self.retained_bytes.checked_sub(other.retained_bytes)?, disk_bytes: self.disk_bytes.checked_sub(other.disk_bytes)?, worker_jobs: self.worker_jobs.checked_sub(other.worker_jobs)?, @@ -274,6 +294,7 @@ impl ResourceCost { self.active_cells <= limit.active_cells && self.resident_bytes <= limit.resident_bytes && self.file_descriptors <= limit.file_descriptors + && self.publication_file_descriptors <= limit.publication_file_descriptors && self.retained_bytes <= limit.retained_bytes && self.disk_bytes <= limit.disk_bytes && self.worker_jobs <= limit.worker_jobs @@ -682,6 +703,49 @@ mod tests { assert_eq!(cost.with_file_descriptors(0).file_descriptors(), 0); } + #[test] + fn publication_descriptor_headroom_cannot_admit_cell_or_reader_handles() { + let ledger = ResourceLedger::new( + ResourceCost::active_cell() + .with_publication_file_descriptors(3) + .with_retained_bytes(16), + ); + let active = ledger.try_reserve(ResourceCost::active_cell()).unwrap(); + assert!( + ledger + .try_reserve(ResourceCost::zero().with_file_descriptors(1)) + .is_err() + ); + let pins = ledger + .try_reserve( + ResourceCost::zero() + .with_publication_file_descriptors(3) + .with_retained_bytes(16), + ) + .unwrap(); + assert_eq!( + ledger.snapshot().unwrap().used.file_descriptors(), + ACTIVE_CELL_FILE_DESCRIPTORS + 3 + ); + assert!( + ledger + .try_reserve(ResourceCost::zero().with_publication_file_descriptors(1)) + .is_err() + ); + assert!( + ledger + .try_reserve(ResourceCost::zero().with_retained_bytes(1)) + .is_err() + ); + drop(pins); + let pins = ledger + .try_reserve(ResourceCost::zero().with_publication_file_descriptors(3)) + .unwrap(); + drop(active); + drop(pins); + assert_eq!(ledger.snapshot().unwrap().used, ResourceCost::zero()); + } + #[test] fn concurrent_reservations_release_to_the_same_baseline() { let ledger = ResourceLedger::new( diff --git a/crates/cellule-runtime/src/follower/grant/tests.rs b/crates/cellule-runtime/src/follower/grant/tests.rs index e046c24c..b71e0ac1 100644 --- a/crates/cellule-runtime/src/follower/grant/tests.rs +++ b/crates/cellule-runtime/src/follower/grant/tests.rs @@ -484,3 +484,32 @@ async fn only_fresh_grant_coverage_can_release_native_prefix_records() { ); assert!(f.backend.requests().is_empty()); } + +#[tokio::test] +async fn enrollment_coverage_ahead_of_queued_frames_does_not_skip_the_requested_witness() { + let mut f = Fixture::new().await; + let leader = f + .directory + .load(f.peer.session, NOW) + .await + .unwrap() + .unwrap(); + f.directory + .advance_log_coverage(&leader, 6, NOW) + .await + .unwrap(); + // The source queued sequence 1 while its object floor was still zero. + // Its exact batch receipt requires that witness even if publication wins + // the race before grant issuance. Fresh authority permits a lower floor. + let grant = f.grant(1).await.unwrap(); + assert_eq!(grant.covered_through(), 0); + let frame = f.frame(1); + let receipt = f.append(&grant, vec![frame.clone()]).await.unwrap(); + assert_eq!(receipt.base_sequence, 1); + assert_eq!(receipt.durable_through, 1); + f.follower.seal(f.peer.session, 2).await.unwrap(); + assert_eq!( + f.follower.read_tail(f.peer.session, 2, 1).await.unwrap(), + vec![frame] + ); +} diff --git a/crates/cellule-runtime/src/node/append_grant/mod.rs b/crates/cellule-runtime/src/node/append_grant/mod.rs index 07ceafb5..bf5c514b 100644 --- a/crates/cellule-runtime/src/node/append_grant/mod.rs +++ b/crates/cellule-runtime/src/node/append_grant/mod.rs @@ -50,6 +50,7 @@ impl NodeAppendGrant { || grant.last_sequence().checked_sub(grant.first_sequence()) != Some(APPEND_GRANT_SEQUENCES - 1) || grant.first_sequence() == 0 + || grant.covered_through() >= grant.first_sequence() || grant.log_epoch() == 0 || !grant.members().contains(&receiver.node) { @@ -245,7 +246,10 @@ impl NodeDirectory { epoch, first, last, - log.tiered_through(), + // Publication may pass frames already queued by the source. Its + // exact batch receipt still needs the first requested witness. + // Fresh authority permits less pruning; it never permits more. + log.tiered_through().min(first - 1), now_ms as u64, expires as u64, ] { diff --git a/crates/cellule-runtime/src/publication/shared/mod.rs b/crates/cellule-runtime/src/publication/shared/mod.rs index c51e2716..f2fabca2 100644 --- a/crates/cellule-runtime/src/publication/shared/mod.rs +++ b/crates/cellule-runtime/src/publication/shared/mod.rs @@ -119,7 +119,7 @@ impl SharedPublication { .ok_or(Error::Capacity("shared publication memory"))?; let cost = ResourceCost::zero() .with_retained_bytes(memory) - .with_file_descriptors(cuts.segments.len() + 2); + .with_publication_file_descriptors(cuts.segments.len() + 2); let memory = match self.resources.try_reserve(cost) { Ok(memory) => memory, // Sharing is optional representation reduction. Never wait for diff --git a/crates/cellule-runtime/src/publication/shared/tests.rs b/crates/cellule-runtime/src/publication/shared/tests.rs index 6f9aa5e3..0f127244 100644 --- a/crates/cellule-runtime/src/publication/shared/tests.rs +++ b/crates/cellule-runtime/src/publication/shared/tests.rs @@ -7,10 +7,65 @@ fn resources(memory: usize) -> ResourceLedger { ResourceLedger::new( ResourceCost::zero() .with_retained_bytes(memory) - .with_file_descriptors(64), + .with_publication_file_descriptors(64), ) } +#[tokio::test] +async fn fully_resident_worker_pool_can_admit_a_shared_capture() { + let directory = tempfile::tempdir().unwrap(); + let pool = crate::SqlWorkerPool::new(1, 1).unwrap(); + let host = Host::default().with_local_disk_budget(cellule_ltx::DiskBudget::new(8 << 20)); + let runtime = crate::CellRuntime::new_with_replica_host( + pool.clone(), + 4 << 20, + crate::SessionId::from_bytes([9; 16]), + host.clone(), + ) + .unwrap(); + let ledger = pool.resource_ledger(); + let active = ledger.try_reserve(ResourceCost::active_cell()).unwrap(); + let coordinator = SharedPublication::new(ledger.clone(), Default::default()); + let replica = CellReplica::new( + CellStorageLayout::new( + Store::new(Arc::new(InMemory::new())), + Path::from("resident-shared"), + [3; 16], + ), + [1; 32], + [1; 16], + Limits::default(), + ) + .unwrap() + .with_host(host.clone()); + let mut db = + Db::open_with_host(&directory.path().join("source"), Limits::default(), host).unwrap(); + db.transaction(|tx| tx.execute_batch("CREATE TABLE t(v); INSERT INTO t VALUES(7)")) + .unwrap(); + let cuts = db.capture().unwrap(); + let prepared = coordinator + .submit( + &replica, + &cuts, + directory.path().to_owned(), + coordinator.admit().await.unwrap(), + ) + .await + .unwrap(); + assert!( + prepared.is_some(), + "resident Cell descriptors must not consume node publication headroom" + ); + drop(prepared); + coordinator.shutdown().await.unwrap(); + drop(cuts); + drop(db); + drop(replica); + drop(active); + runtime.shutdown().await.unwrap(); + assert_eq!(ledger.snapshot().unwrap().used, ResourceCost::zero()); +} + #[tokio::test] async fn minimum_host_permits_drain_shared_work_after_a_sibling_waiter_cancels() { let directory = tempfile::tempdir().unwrap(); @@ -49,7 +104,7 @@ async fn minimum_host_permits_drain_shared_work_after_a_sibling_waiter_cancels() .try_reserve( ResourceCost::zero() .with_retained_bytes(512 << 10) - .with_file_descriptors(3), + .with_publication_file_descriptors(3), ) .unwrap(); let (reply, response) = oneshot::channel(); diff --git a/crates/cellule-runtime/tests/runtime/lifecycle/durability/admission.rs b/crates/cellule-runtime/tests/runtime/lifecycle/durability/admission.rs index bfbb1a98..57f09ac4 100644 --- a/crates/cellule-runtime/tests/runtime/lifecycle/durability/admission.rs +++ b/crates/cellule-runtime/tests/runtime/lifecycle/durability/admission.rs @@ -419,6 +419,7 @@ async fn node_byte_reservation_rejects_overcommit_and_releases_capacity() { assert_eq!( full.file_descriptor_capacity(), ACTIVE_CELL_FILE_DESCRIPTORS + + cellule_runtime::fleet::resource::PUBLICATION_FILE_DESCRIPTORS ); assert_eq!(full.retained_bytes(), 1_024); assert_eq!(full.retained_capacity_bytes(), 1_024); @@ -436,6 +437,7 @@ async fn node_byte_reservation_rejects_overcommit_and_releases_capacity() { assert_eq!( empty.file_descriptor_capacity(), ACTIVE_CELL_FILE_DESCRIPTORS + + cellule_runtime::fleet::resource::PUBLICATION_FILE_DESCRIPTORS ); assert_eq!(empty.retained_bytes(), 0); assert_eq!(empty.local_disk_reserved_bytes(), 0); diff --git a/crates/cellule-runtime/tests/runtime/lifecycle/residency.rs b/crates/cellule-runtime/tests/runtime/lifecycle/residency.rs index 24e76030..8c83e267 100644 --- a/crates/cellule-runtime/tests/runtime/lifecycle/residency.rs +++ b/crates/cellule-runtime/tests/runtime/lifecycle/residency.rs @@ -843,6 +843,7 @@ async fn runtime_stats_follow_active_cell_lifecycle() { assert_eq!( runtime.stats().file_descriptor_capacity(), 10 * ACTIVE_CELL_FILE_DESCRIPTORS + + cellule_runtime::fleet::resource::PUBLICATION_FILE_DESCRIPTORS ); handle.drain().await.unwrap(); assert_eq!(runtime.stats().active_cells(), 0); diff --git a/docs/write-performance-delivery.md b/docs/write-performance-delivery.md index 5a5d7296..e2fda382 100644 --- a/docs/write-performance-delivery.md +++ b/docs/write-performance-delivery.md @@ -67,7 +67,33 @@ boundary/layout/document/SQL-peer/Python gates. The write-proof model checked checks validate the implementation contracts; they do not qualify throughput, physical-device durability or bundle-based bucket acknowledgments. -## Evidence +## Measured-path corrections + +The first M2/M3 Docker run exposed two implementation faults before any gain +could be qualified. At 1,000 resident Cells, ordinary handles consumed the +entire descriptor ledger, so all 29,999 measured shared submissions fell back. +Publication now has dedicated bounded descriptor credit. Cell and reader +admission cannot borrow it; aggregate usage includes both classes, and drain +must return all credit. + +Fresh enrollment could also observe object coverage ahead of the first queued +frame. Signing that higher pruning floor let the receiver skip the frame the +shipper required as an exact witness. Grants now sign the lesser of fresh +coverage and `first_sequence - 1`; verification rejects a floor inside the +authorized window. The retained run had no follower proof advancement and +inactive Fleet shipping, so its zero enrollment GETs are a failure diagnostic. +Reports now require healthy Fleet frontiers throughout the window and actual +follower-proof advancement, separately from publication debt stability. + +CI's four paired routing repetitions found 14–23% lower command throughput in +the first implementation; query rates stayed within approximately 4% of +baseline. Shared upload created an additional synced temporary file. Its writer +now closes before the exact length/digest-verified object upload, without a +local durability barrier for that disposable source. Native captures and +follower logs retain their barriers. The contribution of this change requires +a fresh routing comparison; the earlier failure remains evidence. + +## Historical packed-implementation evidence The fixture is **SQL application parity**, not the user's bounded KV workload: 1,000 Cells, 96-byte values, INSERT plus in-command SELECT, and a two-hour @@ -76,7 +102,7 @@ ceilings and tmpfs state, RustFS has 2 CPUs/2 GiB, and the client has 4 CPUs/4 G on one 8-CPU/16-GiB Linux VM. Summed CPU ceilings exceed VM capacity. This profile qualifies neither device persistence nor independent-node isolation. -Latest main is `18eff0f7af47fac09b993157bb444e582072d7cf`, which merged PR 65. +The earlier baseline is `18eff0f7af47fac09b993157bb444e582072d7cf`, which merged PR 65. Its production source matches the audited `397f500a` foundation; three additional main files are design documents. The baseline explicitly overlays measurement hooks. Celld is v0.6.1, `f2bf648663a610eefde71f3547ad61e9b896b1f0`, using the diff --git a/scripts/perf/report.py b/scripts/perf/report.py index 0f01cfaf..4e295f52 100644 --- a/scripts/perf/report.py +++ b/scripts/perf/report.py @@ -229,6 +229,34 @@ def stability_report(directory): 'sample_elapsed_seconds': times, 'series': series, 'slopes_per_second': slopes, 'frontiers': frontiers, 'frontier_valid': frontier_valid} +def fleet_mode_report(directory): + """Zero authorization cost is meaningless after shipping falls back to roots.""" + paths = sorted(directory.glob('metrics-window-*.json')) + if not (directory / 'metrics-window-start.json').exists() or not (directory / 'metrics-window-end.json').exists(): + return {'available': False, 'pass': False, 'reason': 'fleet window observations missing'} + frontiers = [] + for path in paths: + sample = read(path) + if not sample or not isinstance(sample[0].get('metrics'), dict): + return {'available': False, 'pass': False, 'reason': 'fleet window owner observation missing'} + frontier = sample[0]['metrics'].get('node_log_progress') + if not isinstance(frontier, dict): + return {'available': False, 'pass': False, 'reason': 'fleet window frontier missing'} + frontiers.append(frontier) + passed = all('error' not in frontier and frontier.get('fleet_active') is True + and frontier.get('fenced') is False and frontier.get('rotating') is False + for frontier in frontiers) + first = read(directory / 'metrics-window-start.json')[0]['metrics']['node_log_progress'] + last = read(directory / 'metrics-window-end.json')[0]['metrics']['node_log_progress'] + proven_start, proven_end = first.get('follower_proven_through'), last.get('follower_proven_through') + advancing = (isinstance(proven_start, int) and isinstance(proven_end, int) + and proven_end > proven_start) + passed = passed and advancing + return {'available': True, 'pass': passed, 'frontiers': frontiers, + 'follower_proof_advanced': advancing, + 'reason': None if passed else 'follower shipping inactive, fenced, rotating or proof did not advance'} + + def case_report(directory): case, summary = (read(directory / 'case.json'), read(directory / 'summary.json')) failures = [] @@ -276,10 +304,13 @@ def case_report(directory): metrics = {'available': False, 'error': str(error)} point_failures.append('metric evidence invalid') writes = data['writes'] + fleet_mode = fleet_mode_report(path.parent) if case['system'] == 'cellule' and case['durability'] == 'fleet' else {'available': False, 'pass': False, 'reason': 'not a Cellule Fleet window'} + if case['system'] == 'cellule' and case['durability'] == 'fleet' and not fleet_mode['pass']: + point_failures.append('active follower durability unverified') stability = stability_report(path.parent) if case['system'] == 'cellule' else {'available': False, 'pass': False, 'reason': 'celld publication age not exposed by this fixture'} load_contract = {key: config.get(key) for key in ('cells', 'concurrency', 'queue_capacity', 'write_offset', 'seconds', 'warmup_seconds', 'hot_read_cells')} - points.append({'offered_writes_per_second': config['write_rate'], 'offered_reads_per_second': config['read_rate'], 'successful_writes_per_second': writes['successful_requests_per_second'], 'successful_reads_per_second': data['reads']['successful_requests_per_second'], 'load_contract': load_contract, 'logical_value_bytes_per_second': writes['successful_requests_per_second'] * 96, 'writes': writes, 'reads': data['reads'], 'window_metrics': metrics, 'window_cost': cost_report(metrics, writes['successes_in_window']), 'publication_stability': stability, 'delivery_latency_audit_pass': not point_failures, 'failures': point_failures}) + points.append({'offered_writes_per_second': config['write_rate'], 'offered_reads_per_second': config['read_rate'], 'successful_writes_per_second': writes['successful_requests_per_second'], 'successful_reads_per_second': data['reads']['successful_requests_per_second'], 'load_contract': load_contract, 'logical_value_bytes_per_second': writes['successful_requests_per_second'] * 96, 'writes': writes, 'reads': data['reads'], 'window_metrics': metrics, 'window_cost': cost_report(metrics, writes['successes_in_window']), 'fleet_mode': fleet_mode, 'publication_stability': stability, 'delivery_latency_audit_pass': not point_failures, 'failures': point_failures}) overload = summary.get('overload', {}) recovery = overload.get('recovery') recovery_failures = (delivery_failures(recovery, case['durability']) if recovery else ['recovery phase missing']) + failures diff --git a/scripts/tests/test_perf_report.py b/scripts/tests/test_perf_report.py index f177a933..a0b1faaf 100644 --- a/scripts/tests/test_perf_report.py +++ b/scripts/tests/test_perf_report.py @@ -11,6 +11,54 @@ class DeliveryGateTests(unittest.TestCase): + def test_inactive_fleet_cannot_qualify_with_zero_publication_debt(self): + with tempfile.TemporaryDirectory() as temporary: + directory = Path(temporary) + names = ['metrics-window-start.json', 'metrics-window-minute-1.json', + 'metrics-window-minute-2.json', 'metrics-window-end.json'] + for index, name in enumerate(names): + sample = [{'metrics': {'schema_version': 2, 'sample_session': 'owner', + 'sample_elapsed_ns': index * 60 * 10**9, + 'runtime': {'unpublished_node_log_bytes': 0}, + 'publication_progress': {'oldest_unpublished_ms': None, + 'pending_publications': 0, 'retained_capture_bytes': 0}, + 'node_log_progress': {'fleet_active': False, 'fenced': False, + 'rotating': False, 'tiered_through': 0, + 'issued_through': 0, 'follower_proven_through': 0}}}] + (directory / name).write_text(json.dumps(sample)) + self.assertTrue(REPORT.stability_report(directory)['pass']) + self.assertFalse(REPORT.fleet_mode_report(directory)['pass']) + + def test_active_fleet_requires_healthy_frontiers_throughout_the_window(self): + with tempfile.TemporaryDirectory() as temporary: + directory = Path(temporary) + healthy = {'fleet_active': True, 'fenced': False, 'rotating': False} + paths = [directory / f'metrics-window-{label}.json' + for label in ('start', 'minute-1', 'end')] + for index, path in enumerate(paths): + path.write_text(json.dumps([{'metrics': {'node_log_progress': + dict(healthy, follower_proven_through=index)}}])) + self.assertTrue(REPORT.fleet_mode_report(directory)['pass']) + paths[-1].write_text(json.dumps([{'metrics': {'node_log_progress': + dict(healthy, follower_proven_through=0)}}])) + self.assertFalse(REPORT.fleet_mode_report(directory)['pass']) + paths[-1].write_text(json.dumps([{'metrics': {'node_log_progress': + dict(healthy, follower_proven_through=2)}}])) + for bad in (dict(healthy, fenced=True), dict(healthy, rotating=True), + dict(healthy, error='lease expired')): + paths[1].write_text(json.dumps([{'metrics': {'node_log_progress': bad}}])) + self.assertFalse(REPORT.fleet_mode_report(directory)['pass']) + + def test_missing_fleet_frontiers_cannot_qualify(self): + with tempfile.TemporaryDirectory() as temporary: + directory = Path(temporary) + self.assertFalse(REPORT.fleet_mode_report(directory)['available']) + for label in ('start', 'end'): + (directory / f'metrics-window-{label}.json').write_text('[{"metrics": {}}]') + result = REPORT.fleet_mode_report(directory) + self.assertFalse(result['available']) + self.assertFalse(result['pass']) + def test_provider_oom_after_measurement_is_retained_as_a_cold_failure(self): with tempfile.TemporaryDirectory() as temporary: directory = Path(temporary) From d351d8866521be6e8118eaea40fcf1af75bc59d7 Mon Sep 17 00:00:00 2001 From: forhappy Date: Wed, 7 Oct 2026 02:51:48 -0700 Subject: [PATCH 003/102] Preserve Fleet mode and evaluator identity in comparison evidence --- docs/write-performance-delivery.md | 2 +- scripts/perf/compare.py | 1 + scripts/perf/report.py | 2 +- 3 files changed, 3 insertions(+), 2 deletions(-) diff --git a/docs/write-performance-delivery.md b/docs/write-performance-delivery.md index e2fda382..27849aa2 100644 --- a/docs/write-performance-delivery.md +++ b/docs/write-performance-delivery.md @@ -86,7 +86,7 @@ Reports now require healthy Fleet frontiers throughout the window and actual follower-proof advancement, separately from publication debt stability. CI's four paired routing repetitions found 14–23% lower command throughput in -the first implementation; query rates stayed within approximately 4% of +the first implementation; steady query rates stayed within approximately 4% of baseline. Shared upload created an additional synced temporary file. Its writer now closes before the exact length/digest-verified object upload, without a local durability barrier for that disposable source. Native captures and diff --git a/scripts/perf/compare.py b/scripts/perf/compare.py index 70c13655..3f47ce58 100644 --- a/scripts/perf/compare.py +++ b/scripts/perf/compare.py @@ -112,6 +112,7 @@ def compare(matrix): 'delivery_latency_audit_pass': all(point['delivery_latency_audit_pass'] for point in samples), 'failures': [point['failures'] for point in samples], 'publication_stability': [point['publication_stability'] for point in samples], + 'fleet_mode': [point.get('fleet_mode') for point in samples], 'provider_cost': [point['window_cost'] for point in samples], } comparisons.append({'offered_writes_per_second': key[0], diff --git a/scripts/perf/report.py b/scripts/perf/report.py index 4e295f52..fca0d899 100644 --- a/scripts/perf/report.py +++ b/scripts/perf/report.py @@ -314,7 +314,7 @@ def case_report(directory): overload = summary.get('overload', {}) recovery = overload.get('recovery') recovery_failures = (delivery_failures(recovery, case['durability']) if recovery else ['recovery phase missing']) + failures - return {'schema_version': 1, 'provider_health': health, 'provider_lifecycle': lifecycle, 'case': summary['case'], 'system': case['system'], 'durability': case['durability'], 'workload': 'sql-ledger-96', 'seconds': case['seconds'], 'warmup_seconds': case['warmup_seconds'], 'build_manifest_sha256': summary['build_manifest_sha256'], 'completed': summary['completed'], 'expected_acknowledged_rows': expected_acks, 'acknowledgement_count_reconciled': expected_acks is not None and summary.get('acknowledged_rows') == expected_acks, 'points': points, 'failures': failures, 'overload': {'available': bool(overload), 'reference_capacity': overload.get('reference_capacity'), 'reference_requires_paired_qualification': True, 'recovery_30_second_delivery_pass': bool(recovery) and not recovery_failures, 'recovery_failures': recovery_failures, 'drain_seconds': summary.get('drain_seconds'), 'safe_refusals_before_sql_verified': False}, 'qualification_pass': False, 'qualification_unverified': ['three paired repetitions', 'A/A variance', 'sustained debt slopes and age', 'read-only and mixed guardrails', 'safe overload refusals and qualified reference capacity']} + return {'schema_version': 1, 'reporter_sha256': hashlib.sha256(Path(__file__).read_bytes()).hexdigest(), 'provider_health': health, 'provider_lifecycle': lifecycle, 'case': summary['case'], 'system': case['system'], 'durability': case['durability'], 'workload': 'sql-ledger-96', 'seconds': case['seconds'], 'warmup_seconds': case['warmup_seconds'], 'build_manifest_sha256': summary['build_manifest_sha256'], 'completed': summary['completed'], 'expected_acknowledged_rows': expected_acks, 'acknowledgement_count_reconciled': expected_acks is not None and summary.get('acknowledged_rows') == expected_acks, 'points': points, 'failures': failures, 'overload': {'available': bool(overload), 'reference_capacity': overload.get('reference_capacity'), 'reference_requires_paired_qualification': True, 'recovery_30_second_delivery_pass': bool(recovery) and not recovery_failures, 'recovery_failures': recovery_failures, 'drain_seconds': summary.get('drain_seconds'), 'safe_refusals_before_sql_verified': False}, 'qualification_pass': False, 'qualification_unverified': ['three paired repetitions', 'A/A variance', 'sustained debt slopes and age', 'read-only and mixed guardrails', 'safe overload refusals and qualified reference capacity']} if __name__ == '__main__': parser = argparse.ArgumentParser(description=__doc__) parser.add_argument('directory', type=Path) From 120e4aa6af9c17c946eacff825423629997829fc Mon Sep 17 00:00:00 2001 From: forhappy Date: Wed, 7 Oct 2026 03:18:20 -0700 Subject: [PATCH 004/102] Delta shared counters without subtracting cumulative summaries --- scripts/perf/report.py | 10 ++++++++-- scripts/tests/test_perf_report.py | 3 +++ 2 files changed, 11 insertions(+), 2 deletions(-) diff --git a/scripts/perf/report.py b/scripts/perf/report.py index fca0d899..b35a1119 100644 --- a/scripts/perf/report.py +++ b/scripts/perf/report.py @@ -4,6 +4,8 @@ import json from pathlib import Path +REPORTER_SHA256 = hashlib.sha256(Path(__file__).read_bytes()).hexdigest() + def read(path): return json.loads(path.read_text()) @@ -136,7 +138,11 @@ def metric_delta(directory): for outcome, count in values.get('outcomes', {}).items(): if sum((family[operation]['outcomes'][outcome] for family in families.values())) != count: raise ValueError(f'unclassified {operation}.{outcome}') - shared = subtract(first.get('shared_publication', {}), last.get('shared_publication', {})) + # Summary percentiles/means can decrease as more work completes. The + # raw shared_queue/shared_upload histograms below supply their deltas. + shared_counters = lambda snapshot: {name: value for name, value in snapshot.get('shared_publication', {}).items() + if name in ('cohorts', 'cells', 'rows', 'bytes', 'failures', 'large_fallbacks', 'pressure_fallbacks')} + shared = subtract(shared_counters(first), shared_counters(last)) output.append({'shared_publication': shared, 'url': a['url'], 'sample_elapsed_ns': subtract(first['sample_elapsed_ns'], last['sample_elapsed_ns']), 'start_request_ms': [a['request_started_ms'], a['request_finished_ms']], 'end_request_ms': [b['request_started_ms'], b['request_finished_ms']], 'storage_families': families, 'storage': totals, 'histograms': {name: histogram_delta(first['histograms'][name], value) for name, value in last['histograms'].items()}, 'publication': subtract({name: first['writes'][name] for name in ('selected_roots', 'materialized_commits', 'publication_failures', 'uploaded_objects', 'uploaded_bytes')}, {name: last['writes'][name] for name in ('selected_roots', 'materialized_commits', 'publication_failures', 'uploaded_objects', 'uploaded_bytes')}), 'runtime_start': first.get('runtime'), 'runtime_end': last.get('runtime')}) return {'available': True, 'endpoints': output} @@ -314,7 +320,7 @@ def case_report(directory): overload = summary.get('overload', {}) recovery = overload.get('recovery') recovery_failures = (delivery_failures(recovery, case['durability']) if recovery else ['recovery phase missing']) + failures - return {'schema_version': 1, 'reporter_sha256': hashlib.sha256(Path(__file__).read_bytes()).hexdigest(), 'provider_health': health, 'provider_lifecycle': lifecycle, 'case': summary['case'], 'system': case['system'], 'durability': case['durability'], 'workload': 'sql-ledger-96', 'seconds': case['seconds'], 'warmup_seconds': case['warmup_seconds'], 'build_manifest_sha256': summary['build_manifest_sha256'], 'completed': summary['completed'], 'expected_acknowledged_rows': expected_acks, 'acknowledgement_count_reconciled': expected_acks is not None and summary.get('acknowledged_rows') == expected_acks, 'points': points, 'failures': failures, 'overload': {'available': bool(overload), 'reference_capacity': overload.get('reference_capacity'), 'reference_requires_paired_qualification': True, 'recovery_30_second_delivery_pass': bool(recovery) and not recovery_failures, 'recovery_failures': recovery_failures, 'drain_seconds': summary.get('drain_seconds'), 'safe_refusals_before_sql_verified': False}, 'qualification_pass': False, 'qualification_unverified': ['three paired repetitions', 'A/A variance', 'sustained debt slopes and age', 'read-only and mixed guardrails', 'safe overload refusals and qualified reference capacity']} + return {'schema_version': 1, 'reporter_sha256': REPORTER_SHA256, 'provider_health': health, 'provider_lifecycle': lifecycle, 'case': summary['case'], 'system': case['system'], 'durability': case['durability'], 'workload': 'sql-ledger-96', 'seconds': case['seconds'], 'warmup_seconds': case['warmup_seconds'], 'build_manifest_sha256': summary['build_manifest_sha256'], 'completed': summary['completed'], 'expected_acknowledged_rows': expected_acks, 'acknowledgement_count_reconciled': expected_acks is not None and summary.get('acknowledged_rows') == expected_acks, 'points': points, 'failures': failures, 'overload': {'available': bool(overload), 'reference_capacity': overload.get('reference_capacity'), 'reference_requires_paired_qualification': True, 'recovery_30_second_delivery_pass': bool(recovery) and not recovery_failures, 'recovery_failures': recovery_failures, 'drain_seconds': summary.get('drain_seconds'), 'safe_refusals_before_sql_verified': False}, 'qualification_pass': False, 'qualification_unverified': ['three paired repetitions', 'A/A variance', 'sustained debt slopes and age', 'read-only and mixed guardrails', 'safe overload refusals and qualified reference capacity']} if __name__ == '__main__': parser = argparse.ArgumentParser(description=__doc__) parser.add_argument('directory', type=Path) diff --git a/scripts/tests/test_perf_report.py b/scripts/tests/test_perf_report.py index a0b1faaf..f3c60191 100644 --- a/scripts/tests/test_perf_report.py +++ b/scripts/tests/test_perf_report.py @@ -210,6 +210,8 @@ def sample(count): "sample_elapsed_ns": count * 1000, "storage_families": {"immutable": {"put": operation}}, "storage_operations": {"put": operation}, + "shared_publication": {"cohorts": count, + "queue": {"mean_ms": 10 / count, "p99_ms": 20 / count}}, "histograms": {"worker": {"resolution_us": 100, "total_ns": count * 200000, "buckets": [0, 0, count]}}, "writes": {"selected_roots": count, @@ -225,6 +227,7 @@ def sample(count): self.assertEqual(endpoint["histograms"]["worker"]["resolution_us"], 100) self.assertEqual(endpoint["histograms"]["worker"]["buckets"], [0, 0, 4]) self.assertEqual(endpoint["publication"]["materialized_commits"], 8) + self.assertEqual(endpoint["shared_publication"], {"cohorts": 4}) bad = sample(5) bad[0]["metrics"]["storage_operations"]["put"] = dict( bad[0]["metrics"]["storage_operations"]["put"], bytes_written=999) From d7dc0d4a6eb28cbaf6cb9efceaa6c1481d5cd348 Mon Sep 17 00:00:00 2001 From: forhappy Date: Wed, 7 Oct 2026 03:26:50 -0700 Subject: [PATCH 005/102] Model bundle selection recovery and delayed materialization --- .github/workflows/coordination-model.yml | 1 + .../cellule-runtime/model/BundleCoverage.cfg | 4 + .../cellule-runtime/model/BundleCoverage.tla | 134 ++++++++++++++++++ .../model/BundleCoverageGC.cfg | 4 + .../model/BundleCoverageGap.cfg | 4 + .../model/BundleCoverageIncomplete.cfg | 4 + .../model/BundleCoverageNodeOnly.cfg | 4 + .../model/BundleCoverageRead.cfg | 4 + crates/cellule-runtime/model/README.md | 21 +++ crates/cellule-runtime/model/check.sh | 12 +- docs/bundle-coverage-proof.md | 62 ++++++++ 11 files changed, 253 insertions(+), 1 deletion(-) create mode 100644 crates/cellule-runtime/model/BundleCoverage.cfg create mode 100644 crates/cellule-runtime/model/BundleCoverage.tla create mode 100644 crates/cellule-runtime/model/BundleCoverageGC.cfg create mode 100644 crates/cellule-runtime/model/BundleCoverageGap.cfg create mode 100644 crates/cellule-runtime/model/BundleCoverageIncomplete.cfg create mode 100644 crates/cellule-runtime/model/BundleCoverageNodeOnly.cfg create mode 100644 crates/cellule-runtime/model/BundleCoverageRead.cfg diff --git a/.github/workflows/coordination-model.yml b/.github/workflows/coordination-model.yml index f3515615..fcf899f7 100644 --- a/.github/workflows/coordination-model.yml +++ b/.github/workflows/coordination-model.yml @@ -37,6 +37,7 @@ jobs: crates/cellule-runtime/model/check.sh fast crates/cellule-runtime/model/check.sh negative crates/cellule-runtime/model/check.sh write-proofs + crates/cellule-runtime/model/check.sh bundle-coverage broad: if: github.event_name != 'pull_request' diff --git a/crates/cellule-runtime/model/BundleCoverage.cfg b/crates/cellule-runtime/model/BundleCoverage.cfg new file mode 100644 index 00000000..c7cdd604 --- /dev/null +++ b/crates/cellule-runtime/model/BundleCoverage.cfg @@ -0,0 +1,4 @@ +CONSTANTS NodeOnly = FALSE SkipComplete = FALSE SkipPrefix = FALSE UnprovedRead = FALSE UnpinnedGC = FALSE +SPECIFICATION Spec +INVARIANTS CellFence ContiguousSelection CompleteSelection ProofSelected ReadProven ColdRecoverable ClosedPrefixFrozen AckRecoverable +CHECK_DEADLOCK FALSE diff --git a/crates/cellule-runtime/model/BundleCoverage.tla b/crates/cellule-runtime/model/BundleCoverage.tla new file mode 100644 index 00000000..49fd2b59 --- /dev/null +++ b/crates/cellule-runtime/model/BundleCoverage.tla @@ -0,0 +1,134 @@ +------------------------ MODULE BundleCoverage ------------------------ +EXTENDS Naturals, Integers, FiniteSets, TLC +CONSTANTS NodeOnly, SkipComplete, SkipPrefix, UnprovedRead, UnpinnedGC +Cells == {1, 2} +Bundles == {1, 2} +VARIABLES open, version, head, epoch, closed, endpoint, + stage, proposedVersion, proposedHead, proposedEpoch, complete, + present, proof, ack, visible, root, badSelection, badPrefix, + badContents +vars == <> +Init == /\ open = TRUE /\ version = 0 /\ head = 0 + /\ epoch = [c \in Cells |-> 1] /\ closed = {} + /\ endpoint = [c \in Cells |-> -1] + /\ stage = [b \in Bundles |-> 0] + /\ proposedVersion = [b \in Bundles |-> -1] + /\ proposedHead = [b \in Bundles |-> -1] + /\ proposedEpoch = [b \in Bundles |-> [c \in Cells |-> 0]] + /\ complete = {} /\ present = {} /\ proof = {} + /\ ack = [c \in Cells |-> 0] /\ visible = [c \in Cells |-> 0] + /\ root = [c \in Cells |-> 0] + /\ badSelection = FALSE /\ badPrefix = FALSE /\ badContents = FALSE + +(* Each bundle carries one exact mutation AND retry outcome for each Cell. + Complete abstracts authentication of all bytes, outcomes and dependencies. + It is a checked input, not an assumption about every uploaded object. *) +Prepare(b) == /\ stage[b] = 0 /\ open /\ closed = {} + /\ (SkipPrefix \/ b = head + 1) + /\ stage' = [stage EXCEPT ![b] = 1] + /\ proposedVersion' = [proposedVersion EXCEPT ![b] = version] + /\ proposedHead' = [proposedHead EXCEPT ![b] = head] + /\ proposedEpoch' = [proposedEpoch EXCEPT ![b] = epoch] + /\ UNCHANGED <> +Upload(b, good) == /\ stage[b] = 1 + /\ stage' = [stage EXCEPT ![b] = 2] + /\ present' = present \cup {b} + /\ complete' = IF good THEN complete \cup {b} ELSE complete + /\ UNCHANGED <> +Select(b) == /\ stage[b] = 2 /\ open /\ b \in present + /\ (SkipComplete \/ b \in complete) + /\ (NodeOnly \/ (proposedVersion[b] = version /\ closed = {} + /\ proposedEpoch[b] = epoch)) + /\ (SkipPrefix \/ (proposedHead[b] = head /\ b = head + 1)) + /\ stage' = [stage EXCEPT ![b] = 3] + /\ head' = b /\ version' = version + 1 + /\ badSelection' = badSelection \/ closed # {} \/ proposedEpoch[b] # epoch + /\ badPrefix' = badPrefix \/ b # head + 1 \/ proposedHead[b] # head + /\ badContents' = badContents \/ b \notin complete + /\ UNCHANGED <> + +(* A lost CAS reply leaves stage=3 and proof absent. Fresh exact reconciliation + may mint the SAME selection; it cannot mint a newly uploaded candidate. + A later selected head does not erase the earlier immutable selection. *) +Reconcile(b) == /\ stage[b] = 3 /\ b \notin proof + /\ b \in present /\ (SkipComplete \/ b \in complete) + /\ proof' = proof \cup {b} + /\ UNCHANGED <> +Ack(c, b) == /\ b \in proof /\ c \notin closed + /\ proposedEpoch[b][c] = epoch[c] /\ ack[c] < b + /\ ack' = [ack EXCEPT ![c] = b] + /\ UNCHANGED <> +Read(c, b) == /\ visible[c] < b /\ (UnprovedRead \/ b \in proof) + /\ proposedEpoch[b][c] = epoch[c] /\ c \notin closed + /\ visible' = [visible EXCEPT ![c] = b] + /\ UNCHANGED <> +Materialize(c, b) == /\ b \in proof /\ b \in present /\ b \in complete + /\ root[c] < b + /\ root' = [root EXCEPT ![c] = b] + /\ UNCHANGED <> +Close(c) == /\ c \notin closed + /\ closed' = closed \cup {c} + /\ endpoint' = [endpoint EXCEPT ![c] = head] + /\ version' = version + 1 + /\ UNCHANGED <> +Transfer(c) == /\ c \in closed /\ epoch[c] = 1 + /\ root[c] >= endpoint[c] + /\ epoch' = [epoch EXCEPT ![c] = 2] + /\ UNCHANGED <> +FenceNode == /\ open /\ open' = FALSE /\ version' = version + 1 + /\ UNCHANGED <> +Collect(b) == /\ b \in present + /\ (UnpinnedGC \/ (\A c \in Cells : root[c] >= b)) + /\ present' = present \ {b} + /\ UNCHANGED <> +Next == (\E b \in Bundles : Prepare(b) \/ Upload(b, TRUE) \/ Upload(b, FALSE) + \/ Select(b) \/ Reconcile(b) \/ Collect(b)) + \/ (\E c \in Cells : Close(c) \/ Transfer(c) + \/ (\E b \in Bundles : Ack(c, b) \/ Read(c, b) \/ Materialize(c, b))) + \/ FenceNode +Spec == Init /\ [][Next]_vars +CellFence == ~badSelection +ContiguousSelection == ~badPrefix +CompleteSelection == ~badContents +ProofSelected == \A b \in proof : stage[b] = 3 +ReadProven == \A c \in Cells : visible[c] = 0 \/ visible[c] \in proof +(* A selected range, including an ACK not yet rooted, stays reconstructible. + The dormant sibling keeps the entire shared object live. Root represents + a separately authenticated checkpoint, including identical retry outcomes. *) +ColdRecoverable == \A b \in Bundles : stage[b] # 3 \/ + (b \in complete /\ (b \in present \/ \A c \in Cells : root[c] >= b)) +ClosedPrefixFrozen == \A c \in closed : head <= endpoint[c] +AckRecoverable == \A c \in Cells : ack[c] = 0 \/ ack[c] \in proof +============================================================================= diff --git a/crates/cellule-runtime/model/BundleCoverageGC.cfg b/crates/cellule-runtime/model/BundleCoverageGC.cfg new file mode 100644 index 00000000..e19a6ebc --- /dev/null +++ b/crates/cellule-runtime/model/BundleCoverageGC.cfg @@ -0,0 +1,4 @@ +CONSTANTS NodeOnly = FALSE SkipComplete = FALSE SkipPrefix = FALSE UnprovedRead = FALSE UnpinnedGC = TRUE +SPECIFICATION Spec +INVARIANT ColdRecoverable +CHECK_DEADLOCK FALSE diff --git a/crates/cellule-runtime/model/BundleCoverageGap.cfg b/crates/cellule-runtime/model/BundleCoverageGap.cfg new file mode 100644 index 00000000..4fab3050 --- /dev/null +++ b/crates/cellule-runtime/model/BundleCoverageGap.cfg @@ -0,0 +1,4 @@ +CONSTANTS NodeOnly = FALSE SkipComplete = FALSE SkipPrefix = TRUE UnprovedRead = FALSE UnpinnedGC = FALSE +SPECIFICATION Spec +INVARIANT ContiguousSelection +CHECK_DEADLOCK FALSE diff --git a/crates/cellule-runtime/model/BundleCoverageIncomplete.cfg b/crates/cellule-runtime/model/BundleCoverageIncomplete.cfg new file mode 100644 index 00000000..7bb72cf5 --- /dev/null +++ b/crates/cellule-runtime/model/BundleCoverageIncomplete.cfg @@ -0,0 +1,4 @@ +CONSTANTS NodeOnly = FALSE SkipComplete = TRUE SkipPrefix = FALSE UnprovedRead = FALSE UnpinnedGC = FALSE +SPECIFICATION Spec +INVARIANT CompleteSelection +CHECK_DEADLOCK FALSE diff --git a/crates/cellule-runtime/model/BundleCoverageNodeOnly.cfg b/crates/cellule-runtime/model/BundleCoverageNodeOnly.cfg new file mode 100644 index 00000000..9320974b --- /dev/null +++ b/crates/cellule-runtime/model/BundleCoverageNodeOnly.cfg @@ -0,0 +1,4 @@ +CONSTANTS NodeOnly = TRUE SkipComplete = FALSE SkipPrefix = FALSE UnprovedRead = FALSE UnpinnedGC = FALSE +SPECIFICATION Spec +INVARIANT CellFence +CHECK_DEADLOCK FALSE diff --git a/crates/cellule-runtime/model/BundleCoverageRead.cfg b/crates/cellule-runtime/model/BundleCoverageRead.cfg new file mode 100644 index 00000000..abe7a483 --- /dev/null +++ b/crates/cellule-runtime/model/BundleCoverageRead.cfg @@ -0,0 +1,4 @@ +CONSTANTS NodeOnly = FALSE SkipComplete = FALSE SkipPrefix = FALSE UnprovedRead = TRUE UnpinnedGC = FALSE +SPECIFICATION Spec +INVARIANT ReadProven +CHECK_DEADLOCK FALSE diff --git a/crates/cellule-runtime/model/README.md b/crates/cellule-runtime/model/README.md index 1655a8d6..6279a97d 100644 --- a/crates/cellule-runtime/model/README.md +++ b/crates/cellule-runtime/model/README.md @@ -71,3 +71,24 @@ This model constrains the implemented [append grant](../docs/append-grants.md) lifecycle and the proposed [bundle authority](../../../docs/bundle-coverage-proof.md). It does not establish a production bundle proof, complete dependency verification, byte-identical recovery, liveness or performance qualification. + +## Bundle selection and delayed materialization + +`check.sh bundle-coverage` explores two ordered shared bundles across two Cell +bindings. Upload, selection and proof reconciliation are separate actions, so +a lost selection reply can be reconciled only from the selected immutable +identity. A Cell closure freezes its selected endpoint in the same authority +transition; transfer requires reconstruction through that endpoint. Node +fencing blocks further selection. + +The positive model checks contiguous selection, complete verified inputs, +proven read visibility and retention of a shared object until both Cells have +authenticated checkpoints. A hot Cell cannot release its dormant sibling's +range. Five broken configurations must expose `CellFence`, +`CompleteSelection`, `ContiguousSelection`, `ReadProven` and `ColdRecoverable`. + +`complete` abstracts successful verification of exact bytes, durable retry +outcomes and all dependencies. `root` abstracts an authenticated exact +checkpoint. The model does not implement or prove those checks, SQLite replay, +cryptography, provider persistence, bounded history or Rust adapter ordering. +No production response is enabled by a passing result. diff --git a/crates/cellule-runtime/model/check.sh b/crates/cellule-runtime/model/check.sh index c4b9456b..3c135d24 100755 --- a/crates/cellule-runtime/model/check.sh +++ b/crates/cellule-runtime/model/check.sh @@ -108,6 +108,16 @@ case "$mode" in run_model WriteProofsNoFence.cfg FrozenTail WriteProofs run_model WriteProofsWallOnly.cfg GrantLifetime WriteProofs ;; + bundle-coverage) + java -cp "$jar" tlc2.TLC -workers 1 -nowarning -noGenerateSpecTE \ + -metadir "$cache_dir/meta-bundle-coverage" \ + -config "$model_dir/BundleCoverage.cfg" "$model_dir/BundleCoverage.tla" + run_model BundleCoverageNodeOnly.cfg CellFence BundleCoverage + run_model BundleCoverageIncomplete.cfg CompleteSelection BundleCoverage + run_model BundleCoverageGap.cfg ContiguousSelection BundleCoverage + run_model BundleCoverageRead.cfg ReadProven BundleCoverage + run_model BundleCoverageGC.cfg ColdRecoverable BundleCoverage + ;; fast) java -cp "$jar" tlc2.TLC -workers 1 -depth 6 -nowarning -noGenerateSpecTE \ -metadir "$cache_dir/meta-fast" \ @@ -131,7 +141,7 @@ case "$mode" in run_model CellCoordinationBrokenRelease.cfg RetainedHasOwner ;; *) - echo "usage: $0 {fast|broad|negative|write-proofs}" >&2 + echo "usage: $0 {fast|broad|negative|write-proofs|bundle-coverage}" >&2 exit 2 ;; esac diff --git a/docs/bundle-coverage-proof.md b/docs/bundle-coverage-proof.md index 150f0785..01394aad 100644 --- a/docs/bundle-coverage-proof.md +++ b/docs/bundle-coverage-proof.md @@ -21,6 +21,28 @@ capture bodies from retained memory while keeping bounded authenticated locators A faster upload without these changes can still accumulate debt and exhaust admission. +The corrected M2/M3 Docker candidate at `d351d886` completed one 300-second +window at 100 offered writes/s: 100 completed writes/s, 15.6-ms scheduled p99, +zero errors/drops/unissued offers, and all 34,001 seed/warmup/window ACKs passed +warm and cold GET/retry audits. Fleet remained active and follower proofs +advanced. This is a diagnostic point, not a qualified capacity or parity claim. + +| Window observation | Value | Remaining work | +| --- | ---: | --- | +| All provider PUT successes/command | 5.4842 | Amortize authority selection, not only payload upload | +| Cell-authority PUT successes/command | 2.2521 | Root lineage and Cell selection remain per-Cell work | +| Immutable PUT successes/command | 2.2499 | Includes payloads, roots and maintenance | +| Node-authority PUT successes/command | 0.9822 | Coverage selection is almost per-command at this arrival rate | +| Fresh owner + receiver enrollment GETs/command | 0.0138 | Grant budget passes in this single window; paired qualification remains required | +| Cells/shared cohort | 1.0023 | At this sparse arrival rate the bounded coordinator mostly flushes singletons | +| Commands/selected Cell root | 1.0000 | Delaying a Cell to collect many commands cannot meet the sparse latency target | + +The strict three-minute trend gate failed: unpublished debt rose by +98.53 bytes/s and oldest debt by 0.152 ms/s in the fitted tail. These slopes +must remain failures in the evidence; one run cannot distinguish sustained +growth from sampling variation. Increasing cohort delay without a measured +latency/debt benefit is not a remedy. + ## Authority and transfer Use the existing canonical node record's mutable log authority for both the @@ -69,6 +91,37 @@ cold-recovery scan. Catalog checkpoints can advance bases only after verified materialization; complete reference inventory must survive every intermediate CAS and cancellation. Its object and metadata costs remain in qualification. +## Production cutover and verification order + +The authority record, Cell binding and proof must change together. Implement +the following vertical slices behind a disabled response gate, then enable the +gate only after the final slice passes. A new wire/record version is an atomic +development-format cutover; an old reader must reject it rather than silently +discarding catalog or coverage fields. + +| Slice | Existing implementation to extend | Required evidence before the next slice | +| --- | --- | --- | +| Canonical authority | Node directory record/CAS and `control/authority/acquisition` | Heartbeat/coverage/closure racing on one CAS preserves every field; live-node transfer freezes the exact old binding; fresh reconciliation after a lost PUT reply | +| Ordered immutable range | Native node frames and node shipper; LTX verified upload/restore | Bounded canonical manifest, exact predecessor and full native commit/outcome bytes; missing, duplicate, overlapping and cross-binding rows rejected | +| Opaque proof and logical endpoint | `node/durability`, actor response/query gates | Upload cannot mint proof; selected exact range can; query and retry visibility stop at the same proven endpoint; lease loss stops proof release | +| Asynchronous roots | Canonical publication and LTX materialization | Root lag does not retain unbounded capture bodies; same bytes and outcomes from base plus selected suffix; materializer errors leave earlier ACKs recoverable | +| Transfer, retirement and inventory | Acquisition history, follower seal/retirement, recovery/backup/retention | Owner death before root materialization, dormant sibling, lost seal, late append and GC grace races preserve every accepted range | +| Enable and qualify | Existing pinned Docker runner/client/auditor | All-ACK warm/cold retry after original nodes are destroyed, three paired stable windows, read guardrails and overload/drain; include checkpoint and maintenance cost | + +For the owner-death case, freeze root materialization, accept commands only +through selected range proofs, destroy the original owner and follower state, +then reconstruct from the bucket. Verify every acknowledged request's returned +value and exact stored retry result. Graceful drain alone does not exercise +this condition. For live-node transfer, hold an upload across binding closure +and require its old CAS to fail before the new writer can execute. + +Before accepting a format, account for the full steady-state cost: immutable +data, manifest/catalog checkpoints, node selection, Cell roots, lineage, +compaction and retries. The 0.05-PUT target applies to their sum. A bundle size +calculation that omits materialization or catalog checkpoints cannot qualify +M4. Checkpoint spacing also needs a bounded cold-restore scan and the same read +guardrail; it is a measured policy decision, not an assumed saving. + ## Protocol evidence `WriteProofs.tla` separates upload, exact selection, binding closure and transfer. @@ -82,3 +135,12 @@ The model assumes complete verified immutable inputs and atomic authority CAS. It does not prove manifests, Rust adapter ordering, SQLite bytes, cryptographic identity, provider semantics, fair completion or the full failure matrix. No bucket response or retention release uses this proposed proof in production. + +`BundleCoverage.tla` extends the bounded model with two shared ranges and two +Cell bindings, separate upload/selection/reconciliation, delayed per-Cell +checkpoints, node fencing and exact closure endpoints. It permits incomplete +uploads but requires their rejection before selection. Its five negative +profiles test node-only selection, unchecked contents, a skipped range, +unproven reads and collection of a selected range before both Cells have +checkpoints. Exact bytes/outcomes and checkpoint authentication remain abstract +inputs; this is protocol evidence, not a production recovery implementation. From 0924dfd0d0538629feaf4669d6774056de77b3a3 Mon Sep 17 00:00:00 2001 From: forhappy Date: Wed, 7 Oct 2026 03:27:41 -0700 Subject: [PATCH 006/102] Specify both bundle upload successor branches --- crates/cellule-runtime/model/BundleCoverage.tla | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/crates/cellule-runtime/model/BundleCoverage.tla b/crates/cellule-runtime/model/BundleCoverage.tla index 49fd2b59..6f04fbe9 100644 --- a/crates/cellule-runtime/model/BundleCoverage.tla +++ b/crates/cellule-runtime/model/BundleCoverage.tla @@ -38,7 +38,7 @@ Prepare(b) == /\ stage[b] = 0 /\ open /\ closed = {} Upload(b, good) == /\ stage[b] = 1 /\ stage' = [stage EXCEPT ![b] = 2] /\ present' = present \cup {b} - /\ complete' = IF good THEN complete \cup {b} ELSE complete + /\ complete' = (IF good THEN complete \cup {b} ELSE complete) /\ UNCHANGED <> From 57196d23b234ddf5bbc1a898caa4d2c9fc87c3c1 Mon Sep 17 00:00:00 2001 From: forhappy Date: Wed, 7 Oct 2026 03:30:24 -0700 Subject: [PATCH 007/102] Make unsafe bundle branch assignments explicit --- crates/cellule-runtime/model/BundleCoverage.tla | 6 +++--- 1 file changed, 3 insertions(+), 3 deletions(-) diff --git a/crates/cellule-runtime/model/BundleCoverage.tla b/crates/cellule-runtime/model/BundleCoverage.tla index 6f04fbe9..36c927c7 100644 --- a/crates/cellule-runtime/model/BundleCoverage.tla +++ b/crates/cellule-runtime/model/BundleCoverage.tla @@ -49,9 +49,9 @@ Select(b) == /\ stage[b] = 2 /\ open /\ b \in present /\ (SkipPrefix \/ (proposedHead[b] = head /\ b = head + 1)) /\ stage' = [stage EXCEPT ![b] = 3] /\ head' = b /\ version' = version + 1 - /\ badSelection' = badSelection \/ closed # {} \/ proposedEpoch[b] # epoch - /\ badPrefix' = badPrefix \/ b # head + 1 \/ proposedHead[b] # head - /\ badContents' = badContents \/ b \notin complete + /\ badSelection' = (badSelection \/ closed # {} \/ proposedEpoch[b] # epoch) + /\ badPrefix' = (badPrefix \/ b # head + 1 \/ proposedHead[b] # head) + /\ badContents' = (badContents \/ b \notin complete) /\ UNCHANGED <> From d30e6f1141f44efda9796ff98c90d9dfcae764c5 Mon Sep 17 00:00:00 2001 From: forhappy Date: Wed, 7 Oct 2026 03:32:26 -0700 Subject: [PATCH 008/102] Document verified write gap and gate serving on node liveness --- .../cellule-runtime/model/BundleCoverage.tla | 4 +- docs/write-performance-delivery.md | 58 +++++++++++++++++-- 2 files changed, 56 insertions(+), 6 deletions(-) diff --git a/crates/cellule-runtime/model/BundleCoverage.tla b/crates/cellule-runtime/model/BundleCoverage.tla index 36c927c7..26a0aba6 100644 --- a/crates/cellule-runtime/model/BundleCoverage.tla +++ b/crates/cellule-runtime/model/BundleCoverage.tla @@ -66,14 +66,14 @@ Reconcile(b) == /\ stage[b] = 3 /\ b \notin proof stage, proposedVersion, proposedHead, proposedEpoch, complete, present, ack, visible, root, badSelection, badPrefix, badContents>> -Ack(c, b) == /\ b \in proof /\ c \notin closed +Ack(c, b) == /\ open /\ b \in proof /\ c \notin closed /\ proposedEpoch[b][c] = epoch[c] /\ ack[c] < b /\ ack' = [ack EXCEPT ![c] = b] /\ UNCHANGED <> -Read(c, b) == /\ visible[c] < b /\ (UnprovedRead \/ b \in proof) +Read(c, b) == /\ open /\ visible[c] < b /\ (UnprovedRead \/ b \in proof) /\ proposedEpoch[b][c] = epoch[c] /\ c \notin closed /\ visible' = [visible EXCEPT ![c] = b] /\ UNCHANGED < Date: Wed, 7 Oct 2026 04:00:28 -0700 Subject: [PATCH 009/102] Reuse canonical packs for singleton publication cohorts --- .../cellule-axum/examples/sql_metrics/mod.rs | 6 ++ crates/cellule-ltx/docs/packed-root-format.md | 6 ++ crates/cellule-ltx/src/replica/shared/mod.rs | 49 +++++++++---- crates/cellule-ltx/tests/cell/roots/shared.rs | 69 +++++++++++++++++++ crates/cellule-runtime/src/fleet/telemetry.rs | 10 +++ .../src/publication/shared/mod.rs | 22 +++--- docs/bundle-coverage-proof.md | 29 ++++++++ docs/write-performance-delivery.md | 11 +++ scripts/perf/report.py | 2 +- scripts/tests/test_perf_report.py | 4 +- 10 files changed, 183 insertions(+), 25 deletions(-) diff --git a/crates/cellule-axum/examples/sql_metrics/mod.rs b/crates/cellule-axum/examples/sql_metrics/mod.rs index 430c1fed..343fca4d 100644 --- a/crates/cellule-axum/examples/sql_metrics/mod.rs +++ b/crates/cellule-axum/examples/sql_metrics/mod.rs @@ -50,6 +50,7 @@ pub(super) struct QueryMetrics { #[derive(Default)] struct SharedMetrics { cohorts: AtomicU64, + singletons: AtomicU64, failures: AtomicU64, cells: AtomicU64, rows: AtomicU64, @@ -325,6 +326,10 @@ impl CellTelemetry for QueryMetrics { } .fetch_add(1, Ordering::Relaxed); } + fn shared_publication_singleton(&self, queue: Duration) { + self.shared.singletons.fetch_add(1, Ordering::Relaxed); + self.shared.queue.observe(queue); + } fn ltx_phase(&self, phase: cellule_ltx::LtxPhase, elapsed: Duration, _succeeded: bool) { use cellule_ltx::LtxPhase; match phase { @@ -509,6 +514,7 @@ impl QueryMetrics { }, "shared_publication": { "cohorts": self.shared.cohorts.load(Ordering::Relaxed), + "singletons": self.shared.singletons.load(Ordering::Relaxed), "failures": self.shared.failures.load(Ordering::Relaxed), "cells": self.shared.cells.load(Ordering::Relaxed), "rows": self.shared.rows.load(Ordering::Relaxed), diff --git a/crates/cellule-ltx/docs/packed-root-format.md b/crates/cellule-ltx/docs/packed-root-format.md index f2db57a7..96bb4373 100644 --- a/crates/cellule-ltx/docs/packed-root-format.md +++ b/crates/cellule-ltx/docs/packed-root-format.md @@ -46,6 +46,12 @@ Objects live beneath `shared/objects/.spack` in the application prefix. Application-owned writer, reader, recovery and backup credentials must allow that shared prefix as well as the existing Cell object prefixes. +A cohort with one input and one coalesced row retains its verified native +capture for the ordinary `.pack` factory. It creates no `.spack` or disposable +shared-upload file. Its root uses `shared: false`; scope, exact bytes and fenced +selection follow the same canonical path. Multi-row cohorts retain shared +publication even when the rows belong to one Cell. + Root descriptors require `shared: true` and `packed: true`, pin the complete object digest and exact body/index extents, and retain the native body digest. Ordinary descriptors require `shared: false`. Inventory, compaction and full diff --git a/crates/cellule-ltx/src/replica/shared/mod.rs b/crates/cellule-ltx/src/replica/shared/mod.rs index 22f1b90a..93b95aae 100644 --- a/crates/cellule-ltx/src/replica/shared/mod.rs +++ b/crates/cellule-ltx/src/replica/shared/mod.rs @@ -107,10 +107,11 @@ impl SharedCaptures { } } -/// One Cell's verified extents in an uploaded shared object. +/// One Cell's verified inputs from a bounded publication cohort. /// -/// The private factory waits for exact upload completion. The runtime must still -/// select a prepared root under that Cell's writer fence before acknowledging. +/// Multi-row inputs contain uploaded shared extents. A singleton retains its +/// native captures for the canonical pack factory. Neither grants authority; +/// the runtime must select a prepared root under the Cell's writer fence. #[derive(Clone)] pub struct SharedAppend { replica: CellReplica, @@ -173,7 +174,9 @@ impl CellReplica { })) } - /// Uploads one file-backed cohort once and returns independently scoped inputs. + /// Uploads a multi-row cohort once and returns independently scoped inputs. + /// A singleton retains its verified native inputs: `prepare_shared` uses the + /// ordinary pack factory without a redundant shared file or scratch upload. /// The caller owns retained-memory admission for row/index tables and the /// bounded upload buffer. Scratch uses the same host ledger as compaction. pub async fn upload_shared( @@ -198,6 +201,17 @@ impl CellReplica { if size > SINGLE_PUT_BYTES || rows == 0 || rows > SHARED_PUBLICATION_ROWS { return Err(LtxError::Limit(crate::LimitKind::CellBundleBytes)); } + if inputs.len() == 1 && rows == 1 { + return Ok(inputs + .into_iter() + .map(|input| SharedAppend { + replica: input.replica, + position: input.position, + segments: input.segments, + original: input.original, + }) + .collect()); + } let host = replica.host.for_scratch(size).await?; let disk = host.reserve_local_disk(size)?; let (cleaned, cleanup) = tokio::sync::oneshot::channel(); @@ -318,7 +332,7 @@ impl CellReplica { result } - /// Prepares a scoped shared append through the canonical root factory. + /// Prepares scoped cohort inputs through the canonical root factory. pub async fn prepare_shared( &self, base: Option<&RootRef>, @@ -354,16 +368,23 @@ impl CellReplica { let inputs = append .segments .iter() - .map(|segment| AppendInput { - info: segment.descriptor.info.clone(), - location: BodyLocation::Shared { - digest: segment.descriptor.object_digest(), - offset: segment.descriptor.offset(), - }, - index: segment.index.clone(), - body: AppendBody::SharedUploaded, + .map(|segment| { + let location = match &segment.body { + AppendBody::SharedUploaded => BodyLocation::Shared { + digest: segment.descriptor.object_digest(), + offset: segment.descriptor.offset(), + }, + AppendBody::Native(_) | AppendBody::Frozen(_) => BodyLocation::Native, + _ => return Err(LtxError::InvalidState("invalid cohort input body")), + }; + Ok(AppendInput { + info: segment.descriptor.info.clone(), + location, + index: segment.index.clone(), + body: segment.body.clone(), + }) }) - .collect(); + .collect::>>()?; replica .prepare_append( base, diff --git a/crates/cellule-ltx/tests/cell/roots/shared.rs b/crates/cellule-ltx/tests/cell/roots/shared.rs index abc557cd..2dc64910 100644 --- a/crates/cellule-ltx/tests/cell/roots/shared.rs +++ b/crates/cellule-ltx/tests/cell/roots/shared.rs @@ -1,5 +1,74 @@ use super::*; +#[tokio::test] +async fn singleton_cohort_uses_canonical_packs_and_identical_native_or_coalesced_roots() { + let directory = tempfile::tempdir().unwrap(); + let store = Store::new(Arc::new(InMemory::new())); + for count in 1..=2_u8 { + let replica = replica(store.clone(), [100 + count; 32], [110 + count; 16]); + let mut db = Db::open( + &directory.path().join(format!("source-{count}")), + Limits::default(), + ) + .unwrap(); + db.transaction(|tx| tx.execute_batch("CREATE TABLE t(v); INSERT INTO t VALUES(7)")) + .unwrap(); + let mut cuts = db.capture().unwrap(); + if count == 2 { + db.transaction(|tx| tx.execute_batch("INSERT INTO t VALUES(11)")) + .unwrap(); + let next = db.capture().unwrap(); + cuts.segments.extend(next.segments); + cuts.position = next.position; + } + let inputs = vec![replica.shared_captures(&cuts).await.unwrap().unwrap()]; + let appends = CellReplica::upload_shared(inputs, directory.path()) + .await + .unwrap(); + assert_eq!(replica.publication_cost().objects, 0); + let cohort = replica + .prepare_shared(None, &appends[0], u64::from(count), 1) + .await + .unwrap(); + let direct = replica + .prepare(None, &cuts, u64::from(count), 1) + .await + .unwrap(); + assert_eq!(cohort.root(), direct.root()); + let objects = replica.reachable_objects(&cohort.root()).await.unwrap(); + assert!( + objects + .iter() + .any(|object| object.kind == CellObjectKind::Packed) + ); + assert!( + !objects + .iter() + .any(|object| object.kind == CellObjectKind::SharedPacked) + ); + let destination = directory.path().join(format!("cohort-{count}")); + let canonical = directory.path().join(format!("canonical-{count}")); + cohort.verified().restore(&destination).await.unwrap(); + direct.verified().restore(&canonical).await.unwrap(); + assert_eq!( + std::fs::read(&destination).unwrap(), + std::fs::read(canonical).unwrap() + ); + let db = rusqlite::Connection::open(destination).unwrap(); + let rows: u8 = db + .query_row("SELECT count(*) FROM t", [], |row| row.get(0)) + .unwrap(); + assert_eq!(rows, count); + } + assert!(std::fs::read_dir(directory.path()).unwrap().all(|entry| { + !entry + .unwrap() + .file_name() + .to_string_lossy() + .contains("shared-publication") + })); +} + #[tokio::test] async fn shared_publication_uploads_once_and_restores_each_exact_cell() { let directory = tempfile::tempdir().unwrap(); diff --git a/crates/cellule-runtime/src/fleet/telemetry.rs b/crates/cellule-runtime/src/fleet/telemetry.rs index 2a3d52c9..a19eb2fc 100644 --- a/crates/cellule-runtime/src/fleet/telemetry.rs +++ b/crates/cellule-runtime/src/fleet/telemetry.rs @@ -243,6 +243,10 @@ pub trait CellTelemetry: Send + Sync { /// Records shared cohort work without Cell identity labels. fn shared_publication(&self, _timing: SharedPublicationTiming) {} + /// Records a lone cohort delegated to canonical native-pack preparation. + /// No shared object or upload is counted; queue age includes the cohort wait. + fn shared_publication_singleton(&self, _queue: Duration) {} + /// Records ordinary-path fallback: `true` means retained/descriptor pressure, /// `false` means the capture cannot fit the small-object representation. fn shared_publication_fallback(&self, _pressure: bool) {} @@ -423,6 +427,12 @@ impl CellTelemetryHandle { } } + pub(crate) fn shared_publication_singleton(&self, queue: Duration) { + if let Some(telemetry) = self.inner.get() { + telemetry.shared_publication_singleton(queue); + } + } + pub(crate) fn node_log_append(&self, acknowledged: bool, bytes: u64) { if let Some(telemetry) = self.inner.get() { telemetry.node_log_append(acknowledged, bytes); diff --git a/crates/cellule-runtime/src/publication/shared/mod.rs b/crates/cellule-runtime/src/publication/shared/mod.rs index f2fabca2..84cc82e0 100644 --- a/crates/cellule-runtime/src/publication/shared/mod.rs +++ b/crates/cellule-runtime/src/publication/shared/mod.rs @@ -268,14 +268,20 @@ async fn upload_cohort( let cells = inputs.len() as u64; let upload_started = Instant::now(); let result = cellule_ltx::CellReplica::upload_shared(inputs, &scratch).await; - telemetry.shared_publication(crate::fleet::telemetry::SharedPublicationTiming { - cells, - rows: rows as u64, - bytes, - queue: age, - upload: upload_started.elapsed(), - succeeded: result.is_ok(), - }); + if cells == 1 && rows == 1 && result.is_ok() { + // The native inputs stay pinned and charged through canonical root + // preparation. This path performs no shared-object upload. + telemetry.shared_publication_singleton(age); + } else { + telemetry.shared_publication(crate::fleet::telemetry::SharedPublicationTiming { + cells, + rows: rows as u64, + bytes, + queue: age, + upload: upload_started.elapsed(), + succeeded: result.is_ok(), + }); + } tracing::debug!( rows, bytes, diff --git a/docs/bundle-coverage-proof.md b/docs/bundle-coverage-proof.md index 01394aad..04b7ef6d 100644 --- a/docs/bundle-coverage-proof.md +++ b/docs/bundle-coverage-proof.md @@ -122,6 +122,32 @@ calculation that omits materialization or catalog checkpoints cannot qualify M4. Checkpoint spacing also needs a bounded cold-restore scan and the same read guardrail; it is a measured policy decision, not an assumed saving. +### Quantified checkpoint constraint + +As a design calculation, assume a manifest is embedded in one immutable data +object and one node-selector PUT selects each 64-command cohort. Those two +requests already cost `2 / 64 = 0.03125` PUTs/command. If ordinary checkpoints +retain three root/lineage/Cell-selection PUTs, the remaining M4 budget requires +`0.03125 + 3 / commands_per_checkpoint <= 0.05`: at least 160 commands per +checkpoint, before compaction, catalog changes or retries. If manifests need a +separate PUT, three requests per 64 commands leave even less checkpoint budget. + +At the uniform 1,000-Cell bucket target of 2,000 commands/s, 160 commands per +Cell span approximately 80 seconds; at 15,000 Fleet commands/s they span about +10.7 seconds. These are inferred cost constraints, not measured performance or +a recommended checkpoint delay. Production must either verify that such lag +has bounded file-backed locators and acceptable cold/sparse-read costs, or +amortize checkpoint authority and lineage work as well. Merely allowing bundle +ACKs while publishing every ordinary root would miss the target. + +For the two-PUT assumption, a cohort must contain more than 40 logical commands +to leave any budget for materialization. At 2,000 commands/s, collecting 40 +commands takes about 20 ms in the uniform arrival model. A flush policy must +budget assembly, upload, selector reconciliation and queueing against the +200-ms bucket p99; extending the existing 1-ms cohort policy without that +measurement would be speculation. Count logical commands separately from +manifest rows because one native capture can cover a command range. + ## Protocol evidence `WriteProofs.tla` separates upload, exact selection, binding closure and transfer. @@ -144,3 +170,6 @@ profiles test node-only selection, unchecked contents, a skipped range, unproven reads and collection of a selected range before both Cells have checkpoints. Exact bytes/outcomes and checkpoint authentication remain abstract inputs; this is protocol evidence, not a production recovery implementation. +At `d30e6f11`, CI completed 21,672 distinct positive states and all five named +counterexamples. The separate append/closure model and its three counterexamples +also passed in that run. diff --git a/docs/write-performance-delivery.md b/docs/write-performance-delivery.md index fa593286..5a5a6892 100644 --- a/docs/write-performance-delivery.md +++ b/docs/write-performance-delivery.md @@ -101,6 +101,17 @@ local durability barrier for that disposable source. Native captures and follower logs retain their barriers. The contribution of this change requires a fresh routing comparison; the earlier failure remains evidence. +Fresh routing CI at `120e4aa6` still found 11.5–19.8% lower leased command +throughput across four paired runs, despite removing the temporary-file fsync. +The next candidate therefore delegates a one-input/one-row cohort's verified +captures to the canonical native-pack root factory, without constructing, +reading back or cleaning up a shared upload file. Multi-row cohorts still share +one upload. Separate singleton counters preserve actual shared-object cost and +upload timing. The new path's regression test checks exact native/coalesced +roots, byte-identical restore and absence of shared objects; its fresh +verification and measurements remain pending. The `d351d886` results below +must not be attributed to this subsequent change. + ## Corrected shared/grant measurement Candidate `d351d886` and main `831877cf` use separate source-content-isolated diff --git a/scripts/perf/report.py b/scripts/perf/report.py index b35a1119..8c4ea83a 100644 --- a/scripts/perf/report.py +++ b/scripts/perf/report.py @@ -141,7 +141,7 @@ def metric_delta(directory): # Summary percentiles/means can decrease as more work completes. The # raw shared_queue/shared_upload histograms below supply their deltas. shared_counters = lambda snapshot: {name: value for name, value in snapshot.get('shared_publication', {}).items() - if name in ('cohorts', 'cells', 'rows', 'bytes', 'failures', 'large_fallbacks', 'pressure_fallbacks')} + if name in ('cohorts', 'singletons', 'cells', 'rows', 'bytes', 'failures', 'large_fallbacks', 'pressure_fallbacks')} shared = subtract(shared_counters(first), shared_counters(last)) output.append({'shared_publication': shared, 'url': a['url'], 'sample_elapsed_ns': subtract(first['sample_elapsed_ns'], last['sample_elapsed_ns']), 'start_request_ms': [a['request_started_ms'], a['request_finished_ms']], 'end_request_ms': [b['request_started_ms'], b['request_finished_ms']], 'storage_families': families, 'storage': totals, 'histograms': {name: histogram_delta(first['histograms'][name], value) for name, value in last['histograms'].items()}, 'publication': subtract({name: first['writes'][name] for name in ('selected_roots', 'materialized_commits', 'publication_failures', 'uploaded_objects', 'uploaded_bytes')}, {name: last['writes'][name] for name in ('selected_roots', 'materialized_commits', 'publication_failures', 'uploaded_objects', 'uploaded_bytes')}), 'runtime_start': first.get('runtime'), 'runtime_end': last.get('runtime')}) return {'available': True, 'endpoints': output} diff --git a/scripts/tests/test_perf_report.py b/scripts/tests/test_perf_report.py index f3c60191..4daea191 100644 --- a/scripts/tests/test_perf_report.py +++ b/scripts/tests/test_perf_report.py @@ -210,7 +210,7 @@ def sample(count): "sample_elapsed_ns": count * 1000, "storage_families": {"immutable": {"put": operation}}, "storage_operations": {"put": operation}, - "shared_publication": {"cohorts": count, + "shared_publication": {"cohorts": count, "singletons": count * 2, "queue": {"mean_ms": 10 / count, "p99_ms": 20 / count}}, "histograms": {"worker": {"resolution_us": 100, "total_ns": count * 200000, "buckets": [0, 0, count]}}, @@ -227,7 +227,7 @@ def sample(count): self.assertEqual(endpoint["histograms"]["worker"]["resolution_us"], 100) self.assertEqual(endpoint["histograms"]["worker"]["buckets"], [0, 0, 4]) self.assertEqual(endpoint["publication"]["materialized_commits"], 8) - self.assertEqual(endpoint["shared_publication"], {"cohorts": 4}) + self.assertEqual(endpoint["shared_publication"], {"cohorts": 4, "singletons": 8}) bad = sample(5) bad[0]["metrics"]["storage_operations"]["put"] = dict( bad[0]["metrics"]["storage_operations"]["put"], bytes_written=999) From 20e21b7953754151eac6af9b212cafc22c8c4a7a Mon Sep 17 00:00:00 2001 From: forhappy Date: Wed, 7 Oct 2026 04:18:03 -0700 Subject: [PATCH 010/102] Model independent Cell progress across binding closure --- .../cellule-runtime/model/BundleCoverage.tla | 42 ++++++++++++++----- crates/cellule-runtime/model/README.md | 9 +++- docs/bundle-coverage-proof.md | 9 +++- 3 files changed, 46 insertions(+), 14 deletions(-) diff --git a/crates/cellule-runtime/model/BundleCoverage.tla b/crates/cellule-runtime/model/BundleCoverage.tla index 26a0aba6..8ab4899c 100644 --- a/crates/cellule-runtime/model/BundleCoverage.tla +++ b/crates/cellule-runtime/model/BundleCoverage.tla @@ -3,6 +3,7 @@ EXTENDS Naturals, Integers, FiniteSets, TLC CONSTANTS NodeOnly, SkipComplete, SkipPrefix, UnprovedRead, UnpinnedGC Cells == {1, 2} Bundles == {1, 2} +RowCells(b) == IF b = 1 THEN Cells ELSE {2} VARIABLES open, version, head, epoch, closed, endpoint, stage, proposedVersion, proposedHead, proposedEpoch, complete, present, proof, ack, visible, root, badSelection, badPrefix, @@ -11,6 +12,8 @@ vars == <> +CellSelected(c) == IF stage[2] = 3 /\ c \in RowCells(2) THEN 2 + ELSE IF stage[1] = 3 THEN 1 ELSE 0 Init == /\ open = TRUE /\ version = 0 /\ head = 0 /\ epoch = [c \in Cells |-> 1] /\ closed = {} /\ endpoint = [c \in Cells |-> -1] @@ -23,10 +26,11 @@ Init == /\ open = TRUE /\ version = 0 /\ head = 0 /\ root = [c \in Cells |-> 0] /\ badSelection = FALSE /\ badPrefix = FALSE /\ badContents = FALSE -(* Each bundle carries one exact mutation AND retry outcome for each Cell. +(* The first bundle carries both Cells; the second carries only the hot Cell. + Each row carries an exact mutation AND retry outcome for its Cell. Complete abstracts authentication of all bytes, outcomes and dependencies. It is a checked input, not an assumption about every uploaded object. *) -Prepare(b) == /\ stage[b] = 0 /\ open /\ closed = {} +Prepare(b) == /\ stage[b] = 0 /\ open /\ RowCells(b) \cap closed = {} /\ (SkipPrefix \/ b = head + 1) /\ stage' = [stage EXCEPT ![b] = 1] /\ proposedVersion' = [proposedVersion EXCEPT ![b] = version] @@ -44,18 +48,32 @@ Upload(b, good) == /\ stage[b] = 1 ack, visible, root, badSelection, badPrefix, badContents>> Select(b) == /\ stage[b] = 2 /\ open /\ b \in present /\ (SkipComplete \/ b \in complete) - /\ (NodeOnly \/ (proposedVersion[b] = version /\ closed = {} - /\ proposedEpoch[b] = epoch)) + /\ (NodeOnly \/ (proposedVersion[b] = version + /\ RowCells(b) \cap closed = {} + /\ \A c \in RowCells(b) : proposedEpoch[b][c] = epoch[c])) /\ (SkipPrefix \/ (proposedHead[b] = head /\ b = head + 1)) /\ stage' = [stage EXCEPT ![b] = 3] /\ head' = b /\ version' = version + 1 - /\ badSelection' = (badSelection \/ closed # {} \/ proposedEpoch[b] # epoch) + /\ badSelection' = (badSelection \/ RowCells(b) \cap closed # {} + \/ \E c \in RowCells(b) : proposedEpoch[b][c] # epoch[c]) /\ badPrefix' = (badPrefix \/ b # head + 1 \/ proposedHead[b] # head) /\ badContents' = (badContents \/ b \notin complete) /\ UNCHANGED <> +(* A disjoint binding closure invalidates the old CAS, but need not strand the + hot Cell. Reuse the immutable bytes only after fresh row-by-row authorization + and predecessor verification; a closed participating binding cannot rebase. *) +Rebase(b) == /\ stage[b] = 2 /\ open /\ b \in present + /\ proposedVersion[b] # version /\ b = head + 1 + /\ proposedHead[b] = head /\ RowCells(b) \cap closed = {} + /\ (\A c \in RowCells(b) : proposedEpoch[b][c] = epoch[c]) + /\ proposedVersion' = [proposedVersion EXCEPT ![b] = version] + /\ UNCHANGED <> + (* A lost CAS reply leaves stage=3 and proof absent. Fresh exact reconciliation may mint the SAME selection; it cannot mint a newly uploaded candidate. A later selected head does not erase the earlier immutable selection. *) @@ -66,7 +84,7 @@ Reconcile(b) == /\ stage[b] = 3 /\ b \notin proof stage, proposedVersion, proposedHead, proposedEpoch, complete, present, ack, visible, root, badSelection, badPrefix, badContents>> -Ack(c, b) == /\ open /\ b \in proof /\ c \notin closed +Ack(c, b) == /\ open /\ b \in proof /\ c \in RowCells(b) /\ c \notin closed /\ proposedEpoch[b][c] = epoch[c] /\ ack[c] < b /\ ack' = [ack EXCEPT ![c] = b] /\ UNCHANGED <> Read(c, b) == /\ open /\ visible[c] < b /\ (UnprovedRead \/ b \in proof) + /\ c \in RowCells(b) /\ proposedEpoch[b][c] = epoch[c] /\ c \notin closed /\ visible' = [visible EXCEPT ![c] = b] /\ UNCHANGED <> Materialize(c, b) == /\ b \in proof /\ b \in present /\ b \in complete + /\ c \in RowCells(b) /\ root[c] < b /\ root' = [root EXCEPT ![c] = b] /\ UNCHANGED <> Close(c) == /\ c \notin closed /\ closed' = closed \cup {c} - /\ endpoint' = [endpoint EXCEPT ![c] = head] + /\ endpoint' = [endpoint EXCEPT ![c] = CellSelected(c)] /\ version' = version + 1 /\ UNCHANGED <> Collect(b) == /\ b \in present - /\ (UnpinnedGC \/ (\A c \in Cells : root[c] >= b)) + /\ (UnpinnedGC \/ (\A c \in RowCells(b) : root[c] >= b)) /\ present' = present \ {b} /\ UNCHANGED <> Next == (\E b \in Bundles : Prepare(b) \/ Upload(b, TRUE) \/ Upload(b, FALSE) - \/ Select(b) \/ Reconcile(b) \/ Collect(b)) + \/ Select(b) \/ Rebase(b) \/ Reconcile(b) \/ Collect(b)) \/ (\E c \in Cells : Close(c) \/ Transfer(c) \/ (\E b \in Bundles : Ack(c, b) \/ Read(c, b) \/ Materialize(c, b))) \/ FenceNode @@ -128,7 +148,7 @@ ReadProven == \A c \in Cells : visible[c] = 0 \/ visible[c] \in proof The dormant sibling keeps the entire shared object live. Root represents a separately authenticated checkpoint, including identical retry outcomes. *) ColdRecoverable == \A b \in Bundles : stage[b] # 3 \/ - (b \in complete /\ (b \in present \/ \A c \in Cells : root[c] >= b)) -ClosedPrefixFrozen == \A c \in closed : head <= endpoint[c] + (b \in complete /\ (b \in present \/ \A c \in RowCells(b) : root[c] >= b)) +ClosedPrefixFrozen == \A c \in closed : CellSelected(c) <= endpoint[c] AckRecoverable == \A c \in Cells : ack[c] = 0 \/ ack[c] \in proof ============================================================================= diff --git a/crates/cellule-runtime/model/README.md b/crates/cellule-runtime/model/README.md index 6279a97d..388e83cc 100644 --- a/crates/cellule-runtime/model/README.md +++ b/crates/cellule-runtime/model/README.md @@ -74,13 +74,18 @@ byte-identical recovery, liveness or performance qualification. ## Bundle selection and delayed materialization -`check.sh bundle-coverage` explores two ordered shared bundles across two Cell -bindings. Upload, selection and proof reconciliation are separate actions, so +`check.sh bundle-coverage` explores an initial shared bundle and a following hot +Cell range across two bindings. Upload, selection and proof reconciliation are separate actions, so a lost selection reply can be reconciled only from the selected immutable identity. A Cell closure freezes its selected endpoint in the same authority transition; transfer requires reconstruction through that endpoint. Node fencing blocks further selection. +Closing the dormant binding does not freeze unrelated Cell progress. A fresh +row-scoped rebase can reuse uploaded hot-Cell bytes after the old CAS loses to +that closure, while a closed participating binding remains rejected. Frozen +endpoints are per Cell, not the node's later global sequence. + The positive model checks contiguous selection, complete verified inputs, proven read visibility and retention of a shared object until both Cells have authenticated checkpoints. A hot Cell cannot release its dormant sibling's diff --git a/docs/bundle-coverage-proof.md b/docs/bundle-coverage-proof.md index 04b7ef6d..7a66d847 100644 --- a/docs/bundle-coverage-proof.md +++ b/docs/bundle-coverage-proof.md @@ -162,7 +162,7 @@ It does not prove manifests, Rust adapter ordering, SQLite bytes, cryptographic identity, provider semantics, fair completion or the full failure matrix. No bucket response or retention release uses this proposed proof in production. -`BundleCoverage.tla` extends the bounded model with two shared ranges and two +`BundleCoverage.tla` extends the bounded model with two ordered ranges and two Cell bindings, separate upload/selection/reconciliation, delayed per-Cell checkpoints, node fencing and exact closure endpoints. It permits incomplete uploads but requires their rejection before selection. Its five negative @@ -173,3 +173,10 @@ inputs; this is protocol evidence, not a production recovery implementation. At `d30e6f11`, CI completed 21,672 distinct positive states and all five named counterexamples. The separate append/closure model and its three counterexamples also passed in that run. + +The next model revision lets the hot Cell continue after its sibling closes. +Closure freezes the sibling's per-Cell endpoint rather than the node's global +head. Reusing an uploaded hot-Cell range after that CAS conflict requires fresh +authorization of every participating row and the unchanged predecessor; it +cannot reopen or publish rows from the closed binding. Its CI evidence is +tracked separately from the earlier state count. From 75ae439d25219e97d3a7af141238dadc0d3b1019 Mon Sep 17 00:00:00 2001 From: forhappy Date: Wed, 7 Oct 2026 04:26:47 -0700 Subject: [PATCH 011/102] Make independent checkpoint authority assumptions explicit --- crates/cellule-runtime/model/BundleCoverage.tla | 7 ++++++- crates/cellule-runtime/model/README.md | 6 ++++-- 2 files changed, 10 insertions(+), 3 deletions(-) diff --git a/crates/cellule-runtime/model/BundleCoverage.tla b/crates/cellule-runtime/model/BundleCoverage.tla index 8ab4899c..29566bfc 100644 --- a/crates/cellule-runtime/model/BundleCoverage.tla +++ b/crates/cellule-runtime/model/BundleCoverage.tla @@ -99,11 +99,16 @@ Read(c, b) == /\ open /\ visible[c] < b /\ (UnprovedRead \/ b \in proof) stage, proposedVersion, proposedHead, proposedEpoch, complete, present, proof, ack, root, badSelection, badPrefix, badContents>> +(* Abstract a fully independent checkpoint selected by the authority catalog. + A root descriptor still referencing this bundle is NOT such a checkpoint. + Byte verification, dependency rewriting and complete reference inventory + remain production obligations, not properties proved by this abstraction. *) Materialize(c, b) == /\ b \in proof /\ b \in present /\ b \in complete /\ c \in RowCells(b) /\ root[c] < b /\ root' = [root EXCEPT ![c] = b] - /\ UNCHANGED <> diff --git a/crates/cellule-runtime/model/README.md b/crates/cellule-runtime/model/README.md index 388e83cc..728ee821 100644 --- a/crates/cellule-runtime/model/README.md +++ b/crates/cellule-runtime/model/README.md @@ -93,7 +93,9 @@ range. Five broken configurations must expose `CellFence`, `CompleteSelection`, `ContiguousSelection`, `ReadProven` and `ColdRecoverable`. `complete` abstracts successful verification of exact bytes, durable retry -outcomes and all dependencies. `root` abstracts an authenticated exact -checkpoint. The model does not implement or prove those checks, SQLite replay, +outcomes and all dependencies. `root` abstracts an authenticated independent +checkpoint selected in the authority catalog, including release of its old range +references. An ordinary root that still refers to shared bundle data cannot +perform that action. The model does not implement or prove those checks, SQLite replay, cryptography, provider persistence, bounded history or Rust adapter ordering. No production response is enabled by a passing result. From 954cacfb075c91f633bf28dd4f66e0344f354e76 Mon Sep 17 00:00:00 2001 From: forhappy Date: Wed, 7 Oct 2026 05:23:29 -0700 Subject: [PATCH 012/102] Model accepted follower suffix during Cell binding drain --- .github/workflows/coordination-model.yml | 1 + crates/cellule-runtime/model/BindingDrain.cfg | 5 + crates/cellule-runtime/model/BindingDrain.tla | 125 ++++++++++++++++++ .../model/BindingDrainLateIssue.cfg | 4 + .../model/BindingDrainRetire.cfg | 4 + .../model/BindingDrainSelectedOnly.cfg | 4 + crates/cellule-runtime/model/README.md | 21 +++ crates/cellule-runtime/model/check.sh | 10 +- docs/bundle-coverage-proof.md | 86 +++++++++--- docs/write-performance-delivery.md | 59 +++++++-- scripts/perf/README.md | 4 + scripts/perf/report.py | 37 ++++-- scripts/tests/test_perf_report.py | 30 ++++- 13 files changed, 346 insertions(+), 44 deletions(-) create mode 100644 crates/cellule-runtime/model/BindingDrain.cfg create mode 100644 crates/cellule-runtime/model/BindingDrain.tla create mode 100644 crates/cellule-runtime/model/BindingDrainLateIssue.cfg create mode 100644 crates/cellule-runtime/model/BindingDrainRetire.cfg create mode 100644 crates/cellule-runtime/model/BindingDrainSelectedOnly.cfg diff --git a/.github/workflows/coordination-model.yml b/.github/workflows/coordination-model.yml index fcf899f7..2b8ec182 100644 --- a/.github/workflows/coordination-model.yml +++ b/.github/workflows/coordination-model.yml @@ -38,6 +38,7 @@ jobs: crates/cellule-runtime/model/check.sh negative crates/cellule-runtime/model/check.sh write-proofs crates/cellule-runtime/model/check.sh bundle-coverage + crates/cellule-runtime/model/check.sh binding-drain broad: if: github.event_name != 'pull_request' diff --git a/crates/cellule-runtime/model/BindingDrain.cfg b/crates/cellule-runtime/model/BindingDrain.cfg new file mode 100644 index 00000000..a9779e96 --- /dev/null +++ b/crates/cellule-runtime/model/BindingDrain.cfg @@ -0,0 +1,5 @@ +CONSTANTS SelectedOnlyClose = FALSE IgnoreIssuanceClose = FALSE DropUncoveredFollowers = FALSE +SPECIFICATION FairSpec +INVARIANTS IssuedPrefixFrozen ClosedAcceptedCovered TerminalFrozen AckRecoverable TransferredRecoverable SelectedRecoverable OrderedBounds +PROPERTIES ClosingDrains +CHECK_DEADLOCK FALSE diff --git a/crates/cellule-runtime/model/BindingDrain.tla b/crates/cellule-runtime/model/BindingDrain.tla new file mode 100644 index 00000000..3b15c95a --- /dev/null +++ b/crates/cellule-runtime/model/BindingDrain.tla @@ -0,0 +1,125 @@ +------------------------- MODULE BindingDrain ------------------------- +EXTENDS Naturals, Integers +CONSTANTS SelectedOnlyClose, IgnoreIssuanceClose, DropUncoveredFollowers +VARIABLES phase, version, issued, follower, followerPresent, ack, + selected, objectsPresent, root, closeThrough, terminal, + stage, candidate, candidateVersion, candidateBase +vars == <> +Init == /\ phase = "Open" /\ version = 0 /\ issued = 0 + /\ follower = 0 /\ followerPresent = TRUE /\ ack = 0 + /\ selected = 0 /\ objectsPresent = TRUE /\ root = 0 + /\ closeThrough = -1 /\ terminal = -1 + /\ stage = "Empty" /\ candidate = 0 + /\ candidateVersion = -1 /\ candidateBase = -1 + +(* Issue abstracts a complete verified native capture assigned by the ordered + lane. BeginClose requires the actor to quiesce SQL and join dispatched work + before freezing this endpoint; a sampled follower watermark cannot replace + that barrier. Followers represent the complete required native member set. *) +Issue == /\ issued < 2 /\ phase # "Transferred" + /\ (IgnoreIssuanceClose \/ phase = "Open") + /\ issued' = issued + 1 + /\ UNCHANGED <> +Replicate == /\ followerPresent /\ follower < issued + /\ phase \in {"Open", "Closing"} + /\ follower' = follower + 1 + /\ UNCHANGED <> +FleetAck == /\ followerPresent /\ ack < follower + /\ phase \in {"Open", "Closing"} + /\ ack' = ack + 1 + /\ UNCHANGED <> +BeginClose == /\ phase = "Open" + /\ phase' = "Closing" /\ closeThrough' = issued + /\ version' = version + 1 + /\ UNCHANGED <> + +(* Closing rejects new issuance, but permits object coverage of its exact + previously assigned prefix. A late old CAS still loses; rebase verifies both + the frozen range and the unchanged predecessor. No new owner exists yet. *) +CanDrain == phase = "Open" \/ (phase = "Closing" /\ selected < closeThrough) +Prepare == /\ CanDrain /\ stage = "Empty" /\ selected < issued + /\ stage' = "Prepared" /\ candidate' = selected + 1 + /\ candidateVersion' = version /\ candidateBase' = selected + /\ UNCHANGED <> +Upload == /\ stage = "Prepared" /\ stage' = "Uploaded" + /\ UNCHANGED <> +Rebase == /\ CanDrain /\ stage = "Uploaded" + /\ candidate = selected + 1 /\ candidateBase = selected + /\ candidateVersion # version + /\ (phase # "Closing" \/ candidate <= closeThrough) + /\ candidateVersion' = version + /\ UNCHANGED <> +Select == /\ CanDrain /\ stage = "Uploaded" /\ candidate <= issued + /\ candidateVersion = version /\ candidateBase = selected + /\ candidate = selected + 1 + /\ (phase # "Closing" \/ candidate <= closeThrough) + /\ selected' = candidate /\ objectsPresent' = TRUE + /\ stage' = "Empty" /\ version' = version + 1 + /\ UNCHANGED <> +FinishClose == /\ phase = "Closing" + /\ (SelectedOnlyClose \/ selected >= closeThrough) + /\ phase' = "Closed" /\ terminal' = selected + /\ version' = version + 1 + /\ UNCHANGED <> +(* Root abstracts an authenticated independently recoverable checkpoint, + including the exact retry outcomes. Transfer reconstructs that terminal + endpoint before exposing a new writer; ordinary bundle references are not + independent checkpoints. Byte verification remains a production obligation. *) +Checkpoint == /\ objectsPresent /\ root < selected + /\ root' = selected /\ version' = version + 1 + /\ UNCHANGED <> +Transfer == /\ phase = "Closed" /\ root >= terminal + /\ phase' = "Transferred" + /\ UNCHANGED <> +RetireFollower == /\ followerPresent + /\ ((DropUncoveredFollowers /\ phase = "Closing") + \/ (phase \in {"Closed", "Transferred"} + /\ selected >= issued)) + /\ followerPresent' = FALSE + /\ UNCHANGED <> +Collect == /\ objectsPresent /\ root >= selected + /\ objectsPresent' = FALSE + /\ UNCHANGED <> +Next == Issue \/ Replicate \/ FleetAck \/ BeginClose \/ Prepare \/ Upload + \/ Rebase \/ Select \/ FinishClose \/ Checkpoint \/ Transfer + \/ RetireFollower \/ Collect +Spec == Init /\ [][Next]_vars +IssuedPrefixFrozen == phase = "Open" \/ issued <= closeThrough +ClosedAcceptedCovered == phase \notin {"Closed", "Transferred"} \/ selected >= closeThrough +TerminalFrozen == terminal = -1 \/ selected <= terminal +AckRecoverable == ack <= root \/ (objectsPresent /\ ack <= selected) + \/ (followerPresent /\ ack <= follower) +TransferredRecoverable == phase # "Transferred" \/ ack <= root +SelectedRecoverable == selected <= root \/ objectsPresent +OrderedBounds == 0 <= ack /\ ack <= follower /\ follower <= issued + /\ root <= selected /\ selected <= issued +FairSpec == Spec /\ WF_vars(Prepare) /\ WF_vars(Upload) /\ WF_vars(Rebase) + /\ WF_vars(Select) /\ WF_vars(FinishClose) +ClosingDrains == [](phase = "Closing" => <>(phase \in {"Closed", "Transferred"})) +============================================================================= diff --git a/crates/cellule-runtime/model/BindingDrainLateIssue.cfg b/crates/cellule-runtime/model/BindingDrainLateIssue.cfg new file mode 100644 index 00000000..ca816f1e --- /dev/null +++ b/crates/cellule-runtime/model/BindingDrainLateIssue.cfg @@ -0,0 +1,4 @@ +CONSTANTS SelectedOnlyClose = FALSE IgnoreIssuanceClose = TRUE DropUncoveredFollowers = FALSE +SPECIFICATION Spec +INVARIANT IssuedPrefixFrozen +CHECK_DEADLOCK FALSE diff --git a/crates/cellule-runtime/model/BindingDrainRetire.cfg b/crates/cellule-runtime/model/BindingDrainRetire.cfg new file mode 100644 index 00000000..ef349a3c --- /dev/null +++ b/crates/cellule-runtime/model/BindingDrainRetire.cfg @@ -0,0 +1,4 @@ +CONSTANTS SelectedOnlyClose = FALSE IgnoreIssuanceClose = FALSE DropUncoveredFollowers = TRUE +SPECIFICATION Spec +INVARIANT AckRecoverable +CHECK_DEADLOCK FALSE diff --git a/crates/cellule-runtime/model/BindingDrainSelectedOnly.cfg b/crates/cellule-runtime/model/BindingDrainSelectedOnly.cfg new file mode 100644 index 00000000..a26c813a --- /dev/null +++ b/crates/cellule-runtime/model/BindingDrainSelectedOnly.cfg @@ -0,0 +1,4 @@ +CONSTANTS SelectedOnlyClose = TRUE IgnoreIssuanceClose = FALSE DropUncoveredFollowers = FALSE +SPECIFICATION Spec +INVARIANT TransferredRecoverable +CHECK_DEADLOCK FALSE diff --git a/crates/cellule-runtime/model/README.md b/crates/cellule-runtime/model/README.md index 728ee821..6a0c2fe9 100644 --- a/crates/cellule-runtime/model/README.md +++ b/crates/cellule-runtime/model/README.md @@ -99,3 +99,24 @@ references. An ordinary root that still refers to shared bundle data cannot perform that action. The model does not implement or prove those checks, SQLite replay, cryptography, provider persistence, bounded history or Rust adapter ordering. No production response is enabled by a passing result. + +## Accepted Fleet suffix during binding closure + +`check.sh binding-drain` composes follower ACKs with later object selection for +one binding and two ordered commands. A selected-only closure could omit a +command already acknowledged by the native follower path. `BeginClose` first +stops issuance and freezes the complete previously assigned range. Closing may +select verified rows only through that endpoint, under a fresh CAS, before +terminal closure and reconstruction permit transfer. Old uploaded proposals +must rebase their authority version and unchanged predecessor; they cannot add +new commands after the boundary. Retirement preserves every ACK's durable copy. + +The positive configuration checks accepted-prefix coverage, frozen issuance, +recoverable ACKs and a terminal endpoint reconstructed before transfer. Under +weakly fair successful preparation/upload/selection, `ClosingDrains` checks +eventual terminal closure. Three unsafe configurations must expose +`TransferredRecoverable`, `IssuedPrefixFrozen` and `AckRecoverable` respectively. +The model assumes exact complete captures and authenticated independent +checkpoints. It does not prove that Rust joins every accepted SQL/capture job, +that native receivers enforce the barrier, or that providers always complete. +It does not enable a production bundle response. diff --git a/crates/cellule-runtime/model/check.sh b/crates/cellule-runtime/model/check.sh index 3c135d24..ff312e5b 100755 --- a/crates/cellule-runtime/model/check.sh +++ b/crates/cellule-runtime/model/check.sh @@ -118,6 +118,14 @@ case "$mode" in run_model BundleCoverageRead.cfg ReadProven BundleCoverage run_model BundleCoverageGC.cfg ColdRecoverable BundleCoverage ;; + binding-drain) + java -cp "$jar" tlc2.TLC -workers 1 -nowarning -noGenerateSpecTE \ + -metadir "$cache_dir/meta-binding-drain" \ + -config "$model_dir/BindingDrain.cfg" "$model_dir/BindingDrain.tla" + run_model BindingDrainSelectedOnly.cfg TransferredRecoverable BindingDrain + run_model BindingDrainLateIssue.cfg IssuedPrefixFrozen BindingDrain + run_model BindingDrainRetire.cfg AckRecoverable BindingDrain + ;; fast) java -cp "$jar" tlc2.TLC -workers 1 -depth 6 -nowarning -noGenerateSpecTE \ -metadir "$cache_dir/meta-fast" \ @@ -141,7 +149,7 @@ case "$mode" in run_model CellCoordinationBrokenRelease.cfg RetainedHasOwner ;; *) - echo "usage: $0 {fast|broad|negative|write-proofs|bundle-coverage}" >&2 + echo "usage: $0 {fast|broad|negative|write-proofs|bundle-coverage|binding-drain}" >&2 exit 2 ;; esac diff --git a/docs/bundle-coverage-proof.md b/docs/bundle-coverage-proof.md index 7a66d847..33ab59d7 100644 --- a/docs/bundle-coverage-proof.md +++ b/docs/bundle-coverage-proof.md @@ -21,11 +21,21 @@ capture bodies from retained memory while keeping bounded authenticated locators A faster upload without these changes can still accumulate debt and exhaust admission. -The corrected M2/M3 Docker candidate at `d351d886` completed one 300-second -window at 100 offered writes/s: 100 completed writes/s, 15.6-ms scheduled p99, -zero errors/drops/unissued offers, and all 34,001 seed/warmup/window ACKs passed -warm and cold GET/retry audits. Fleet remained active and follower proofs -advanced. This is a diagnostic point, not a qualified capacity or parity claim. +At one command per selected root, the three per-Cell metadata PUTs alone would +require 6,000 PUTs/s at the bucket target, before payloads, node selection or +maintenance. The complete M4 budget there is 100 PUTs/s. This is a cost bound +for that sparse schedule, not extrapolated measured throughput. Fleet can +coalesce more root work because a follower proof may respond earlier, but it +still needs bounded retention and eventual object coverage; a longer root timer +cannot release an unproved capture or hide its debt. + +The corrected M2/M3 Docker candidate at `d351d886` completed three 300-second +windows at 100 offered writes/s. Scheduled p99 was 15.6 / 36.4 / 31.2 ms with +zero errors/drops/unissued offers; every run's 34,001 seed/warmup/window ACKs +passed warm and cold GET/retry audits. Fleet remained active and follower +proofs advanced. All three debt trends failed and total PUT cost was +5.330–5.484 per command. These are diagnostic points, not a qualified capacity +or parity claim. The first window supplies the detailed observations below. | Window observation | Value | Remaining work | | --- | ---: | --- | @@ -54,29 +64,42 @@ read node advertisement would reinstate a fencing race. | Object or capability | Exact meaning | | --- | --- | | Cell binding | Cell/incarnation, writer epoch, exact base root/schema/code, permitted node boot/log epoch, and a unique binding identity pinned by Cell control | -| Immutable binding catalog | Complete active bindings plus terminal closure endpoints; closed binding IDs cannot be re-added | +| Immutable binding catalog | Complete Open/Closing bindings plus terminal Closed endpoints; closed binding IDs cannot be re-added | | Immutable range manifest | Contiguous ordered node ranges, exact Cell bindings and commit/transaction intervals, scoped byte extents and digests, complete native outcome/dependency coverage, predecessor head | | Node selection | Post-upload CAS of both catalog and range head while the original node/log remains Open and every row's binding remains active | | `BundleCoverageProof` | Opaque capability minted only after exact selection reconciliation and complete dependency verification; uploaded bytes cannot construct it | | Materialization | Exact logical endpoint reconstructed from base and selected rows; ordinary Cell root/lineage CAS consumes the same proof | -| Closure endpoint | Complete selected prefix frozen in the same CAS that closes the binding; new Cell ownership cannot execute before reconstructing it | - -A live-node Cell transfer must first close its old binding in that node's -canonical record. A proposed upload holding an earlier catalog/head version -then loses its CAS, reloads, and rejects the closed rows. The closed binding -retains all earlier selected rows and its base until exact reconstruction or -retention proof releases them. Only then can Cell authority select the next -writer. Failed-node recovery first fences the whole original node record and -freezes its selected head before closing individual bindings. This ordering +| Closure endpoint | Closing freezes the complete previously issued range; Closed records its complete selected endpoint; new ownership cannot execute before reconstructing it | + +A live-node Cell transfer must first quiesce SQL, join accepted capture/submission +jobs and close issuance for its old binding in the ordered node lane. A CAS +marks that binding Closing and freezes its complete assigned endpoint. Previously +issued, exactly verified rows may drain through that boundary using fresh +authority; a proposed upload holding the earlier catalog/head version loses +its CAS. Terminal Closed rejects all later rows and retains the complete selected +prefix and base until reconstruction or retention proof releases them. Only then +can Cell authority select the next writer. Failed-node recovery first fences +the whole original node record, seals the complete follower endpoint and +accounts for its accepted suffix beyond the selected head before terminal +closure. This ordering must also govern release, takeover, tombstone, migration, backup and failed shutdown; a lower-level Cell transition cannot bypass it. +The accepted endpoint and selected endpoint are distinct. A Fleet command can +already be acknowledged from native followers above the last selected bucket +range. Freezing only that selected prefix could let a new writer omit a prior +ACK. Do not use a sampled follower watermark as the issuance boundary: an +accepted SQL/capture job may still be assigning its ticket. Join the original +jobs, stop assignment, and retain every old range until exact object coverage +or reconstruction consumes it. If the old node cannot establish this barrier, +fence/recover its complete log before exposing the replacement Cell writer. + ## Atomic integration required before enabling responses | Surface | Required production change and rejection case | | --- | --- | | Node advertisement/log codecs | Catalog/head/materialized frontiers preserved by every heartbeat, enrollment, closure and recovery CAS; reject old formats atomically | -| Cell control and acquisition history | Pin exact binding before SQL; close/freeze it before ownership departure; reject live-node late rows | +| Cell control and acquisition history | Pin exact binding before SQL; quiesce/join/stop issuance, drain its frozen accepted range and reconstruct before ownership departure; reject live-node late rows | | Runtime durability gate | Distinct bundle source and proof; exact ticket/row coverage; no proof from upload alone | | Actor and executor | Advance proven outcome/query endpoint together; retain debt until proof; replace proved capture bodies with bounded immutable locators | | Materializer | Coalesce selected intervals under host budgets; root and schema stay byte-identical; failure cannot invalidate an earlier bundle ACK | @@ -90,6 +113,13 @@ long histories. Merely chaining every per-command manifest leaves an unbounded cold-recovery scan. Catalog checkpoints can advance bases only after verified materialization; complete reference inventory must survive every intermediate CAS and cancellation. Its object and metadata costs remain in qualification. +Recovery needs an authenticated lookup of the chosen binding's selected suffix, +not a scan of every unrelated Cell's data object. Bound and measure index +depth, objects/bytes read, retained locator bytes and reconstruction work before +choosing checkpoint spacing. Public receipts already name a logical Cell +incarnation/command sequence, but the read-replica implementation opens a +materialized root: delayed roots must preserve receipt-bound visibility without +claiming that an older root contains the selected suffix. ## Production cutover and verification order @@ -101,7 +131,7 @@ discarding catalog or coverage fields. | Slice | Existing implementation to extend | Required evidence before the next slice | | --- | --- | --- | -| Canonical authority | Node directory record/CAS and `control/authority/acquisition` | Heartbeat/coverage/closure racing on one CAS preserves every field; live-node transfer freezes the exact old binding; fresh reconciliation after a lost PUT reply | +| Canonical authority | Node directory record/CAS and `control/authority/acquisition` | Heartbeat/coverage/closure racing on one CAS preserves every field; live-node transfer covers the frozen issued endpoint, including prior Fleet ACKs beyond the selected prefix; fresh reconciliation after a lost PUT reply | | Ordered immutable range | Native node frames and node shipper; LTX verified upload/restore | Bounded canonical manifest, exact predecessor and full native commit/outcome bytes; missing, duplicate, overlapping and cross-binding rows rejected | | Opaque proof and logical endpoint | `node/durability`, actor response/query gates | Upload cannot mint proof; selected exact range can; query and retry visibility stop at the same proven endpoint; lease loss stops proof release | | Asynchronous roots | Canonical publication and LTX materialization | Root lag does not retain unbounded capture bodies; same bytes and outcomes from base plus selected suffix; materializer errors leave earlier ACKs recoverable | @@ -170,13 +200,25 @@ profiles test node-only selection, unchecked contents, a skipped range, unproven reads and collection of a selected range before both Cells have checkpoints. Exact bytes/outcomes and checkpoint authentication remain abstract inputs; this is protocol evidence, not a production recovery implementation. -At `d30e6f11`, CI completed 21,672 distinct positive states and all five named +At `75ae439d`, CI completed 44,048 distinct positive states and all five named counterexamples. The separate append/closure model and its three counterexamples -also passed in that run. +also passed in that run. This extends the earlier `d30e6f11` run's 21,672 states. -The next model revision lets the hot Cell continue after its sibling closes. +The current model lets the hot Cell continue after its sibling closes. Closure freezes the sibling's per-Cell endpoint rather than the node's global head. Reusing an uploaded hot-Cell range after that CAS conflict requires fresh authorization of every participating row and the unchanged predecessor; it -cannot reopen or publish rows from the closed binding. Its CI evidence is -tracked separately from the earlier state count. +cannot reopen or publish rows from the closed binding. An independent checkpoint +means a fully authenticated base selected by authority; an ordinary root that +still references the bundle cannot authorize collection. The model does not +prove byte codecs, complete production reference inventory, history/backup +retention, grace boundaries or liveness. + +`BindingDrain.tla` adds the missing Fleet composition: native durable ACKs may +precede object selection. Closing freezes issuance and drains exactly that old +range before terminal closure and reconstruction. Its positive profile also +checks eventual closing under weakly fair successful upload/selection. Three +negative profiles require counterexamples for selected-only transfer, late +issuance and premature follower retirement. Verification is pending separately +from the state counts above. Exact captures/checkpoints and the Rust join/close +barrier remain production obligations, not proofs supplied by this abstraction. diff --git a/docs/write-performance-delivery.md b/docs/write-performance-delivery.md index 5a5a6892..cf8bbe40 100644 --- a/docs/write-performance-delivery.md +++ b/docs/write-performance-delivery.md @@ -38,10 +38,10 @@ collection paths. There is no legacy decoding or automatic migration. | --- | --- | --- | | M0 | Measurement and comparison harness delivered | Three A/A capacity pairs unverified; storage API totals reconcile, but SDK-internal HTTP retries need provider telemetry | | M1 | Packs, inline leaves and bounded compaction spooling delivered | Ordinary append meets four PUTs; composed compaction needs five. Paired cold/sparse-read guardrail unverified | -| M2 | Bounded file-backed shared publication coordinator implemented; exact scope, restore, cancellation, minimum-budget and dormant-sibling retention checks added | Corrected active-Fleet window costs 5.484 PUTs/command and fails the debt trend. Three per-Cell authority PUTs remain; M4 is required | -| M3 | Signed 512-sequence/five-second grants, bounded local registry, lifecycle gate and signed HTTP fixture implemented | Full isolated checks and native lifecycle suite pass; one active-Fleet window costs 0.0138 enrollment GETs/command. Three-repetition gate pending | +| M2 | Bounded file-backed shared publication coordinator implemented; exact scope, restore, cancellation, minimum-budget and dormant-sibling retention checks added | Three corrected active-Fleet windows cost 5.330–5.484 PUTs/command and all fail the debt trend. Three per-Cell authority PUTs remain; M4 is required | +| M3 | Signed 512-sequence/five-second grants, bounded local registry, lifecycle gate and signed HTTP fixture implemented | Full isolated checks and native lifecycle suite pass; three active-Fleet windows cost 0.0138–0.0141 enrollment GETs/command. Target-rate qualification remains unverified | | M4 | Binding/selector and delayed-materialization models plus [authority decision](bundle-coverage-proof.md) delivered | Production bundle proof, atomic transfer/recovery/collection and bundle ACKs not implemented | -| M5 | Matched Fleet/read points and target stress with exact ACK audits delivered | Three repetitions, read/failure/overload matrix and absolute/relative parity unverified | +| M5 | Three paired low-rate Fleet repetitions and earlier read/target stress with exact ACK audits delivered | Publication stability fails; qualified capacity, read/failure/overload matrix and absolute/relative parity remain unverified | ## Shared publication checkpoint @@ -109,9 +109,27 @@ reading back or cleaning up a shared upload file. Multi-row cohorts still share one upload. Separate singleton counters preserve actual shared-object cost and upload timing. The new path's regression test checks exact native/coalesced roots, byte-identical restore and absence of shared objects; its fresh -verification and measurements remain pending. The `d351d886` results below +isolated verification passed 1,862 workspace tests (38 ignored), 58 local LTX +tests, all contributor checks and the complete 382-test serial native lifecycle +suite. The release binary is pinned to `a3787a8d`; all 1,248 Rust/Cargo source +hashes match the verified snapshot. Three matched Docker repetitions and fresh +routing CI are tracked separately. The `d351d886` results below must not be attributed to this subsequent change. +Fresh routing CI for `75ae439d` passed all four paired command comparisons in +both modes against `831877cf`. Median command-throughput ratios are +0.992–1.069 in leased mode and 0.967–1.143 in object-only mode. Steady local and +forwarded query-throughput ratios are 0.958–1.015. These native routing workloads +have different transaction/arrival contracts from the SQL Docker comparison. +Their CI latency limit is two times baseline, not the proposal's 1.2 read limit: +object-only forwarded/local concurrency-one query p99 ratios were 1.614 / 1.241. +Passing routing CI therefore does not pass the M5 read guardrail. + +The same head's workspace CI failed one combined-maintenance lifecycle test with +a nested fleet-action-journal conflict (381/382 passed); a fresh rerun is +pending. The isolated complete native suite passed all 382 on the same production +source. Both results are retained; the CI conflict's cause is not established. + ## Corrected shared/grant measurement Candidate `d351d886` and main `831877cf` use separate source-content-isolated @@ -120,10 +138,30 @@ release builds, with no measurement overlay. Celld is pinned to v0.6.1 the arms. The current profile is the SQL ledger workload below on one shared 8-CPU/16-GiB VM, with tmpfs node state and fresh RustFS volumes. -The first corrected candidate window offered 100 writes/s for 300 seconds after -30 seconds warmup. All original node containers were removed after drain and -before the bucket-only cold audit. Paired repeats are in progress; the following -is one diagnostic window, not a sustainable-capacity or parity claim. +Three paired repetitions offered 100 writes/s for 300 seconds after 30 seconds +warmup, alternating system order. All original node containers were removed +after drain and before each bucket-only cold audit. These are matched diagnostic +points, not a sustainable-capacity or parity claim. + +| System | Window commands/s, repetitions 1 / 2 / 3 | Scheduled p99, ms, repetitions 1 / 2 / 3 | Point delivery/latency result | +| --- | --- | --- | --- | +| Main `831877cf` | 99.997 / 100.000 / 99.997 | 123.2 / 89.3 / 90.0 | All exceed 50 ms; first also loses active Fleet shipping | +| Candidate `d351d886` | 100.000 / 99.987 / 100.000 | 15.6 / 36.4 / 31.2 | All pass; all three publication-debt trends fail | +| celld `f2bf6486` | 100.000 / 99.997 / 100.000 | 11.1 / 15.8 / 14.1 | First/third pass; second has one request-identity validity error | + +Every actual ACK passed warm and cold GET plus exact retry: 34,001 per Cellule +run and 34,001 / 34,000 / 34,001 for celld. The celld error is not an ACK-loss +observation. Its underlying clock/identity cause remains unestablished; no +backdating or policy change is applied. Main's first follower rejected a signed +deadline above its allowed horizon before directory verification; zero final +debt after fallback does not qualify Fleet. The original failed cases remain +in the comparison. + +Candidate PUTs/command were 5.4842 / 5.3301 / 5.3969 and summed fresh enrollment +GETs/command were 0.0138 / 0.0141 / 0.0138. Candidate tail-debt slopes were ++98.53 / +1,588.95 / +12.54 bytes/s; all remain failures. Celld debt age is not +exposed by this fixture, so its stability is unavailable rather than passing. +The first candidate window supplies the phase and cohort details below. | Candidate window/audit | Measured value | | --- | ---: | @@ -141,8 +179,9 @@ is one diagnostic window, not a sustainable-capacity or parity claim. | Final debt / oldest-age slope | +98.53 bytes/s / +0.152 ms/s | Fleet stayed active, unfenced and non-rotating throughout the sampled window; -its follower proof frontier advanced. Delivery, latency and all-ACK audits pass -at this point. The strict publication stability gate fails, and PUT cost is +its follower proof frontier advanced in every repetition. Delivery, latency and +all-ACK audits pass at these points. The strict publication stability gate fails, +and PUT cost is well above M2's 0.25 budget. Near-singleton cohorts show why shared payloads alone cannot amortize sparse per-Cell authority work. The [bundle decision](bundle-coverage-proof.md) describes the remaining atomic diff --git a/scripts/perf/README.md b/scripts/perf/README.md index f69de848..f85b91e2 100644 --- a/scripts/perf/README.md +++ b/scripts/perf/README.md @@ -143,6 +143,10 @@ Cellule stability reports fit the debt and oldest-publication age over the last three one-minute segments. A positive slope fails; missing, late, reset, or fenced observations cannot pass. Native log frontiers remain observations and grant no durability or collection authority. +Fleet write windows also require follower-proof advancement. Read-only windows +may retain the same frontier, but still require active Fleet, one unchanged log +epoch, monotonic valid counters and no fencing/rotation throughout the window. +Their seed ACKs remain subject to the complete warm/cold audit. For a machine-readable comparison, write an external JSON file with `baseline`, `candidate` and `celld` lists of case directories, then run: diff --git a/scripts/perf/report.py b/scripts/perf/report.py index 8c4ea83a..388e6f5b 100644 --- a/scripts/perf/report.py +++ b/scripts/perf/report.py @@ -235,9 +235,12 @@ def stability_report(directory): 'sample_elapsed_seconds': times, 'series': series, 'slopes_per_second': slopes, 'frontiers': frontiers, 'frontier_valid': frontier_valid} -def fleet_mode_report(directory): - """Zero authorization cost is meaningless after shipping falls back to roots.""" - paths = sorted(directory.glob('metrics-window-*.json')) +def fleet_mode_report(directory, *, writes_offered=True): + """Require live Fleet and new proofs for writes, healthy fixed proofs for reads.""" + paths = [directory / 'metrics-window-start.json', + *sorted(directory.glob('metrics-window-minute-*.json'), + key=lambda path: int(path.stem.rsplit('-', 1)[1])), + directory / 'metrics-window-end.json'] if not (directory / 'metrics-window-start.json').exists() or not (directory / 'metrics-window-end.json').exists(): return {'available': False, 'pass': False, 'reason': 'fleet window observations missing'} frontiers = [] @@ -249,18 +252,32 @@ def fleet_mode_report(directory): if not isinstance(frontier, dict): return {'available': False, 'pass': False, 'reason': 'fleet window frontier missing'} frontiers.append(frontier) - passed = all('error' not in frontier and frontier.get('fleet_active') is True + healthy = all('error' not in frontier and frontier.get('fleet_active') is True and frontier.get('fenced') is False and frontier.get('rotating') is False for frontier in frontiers) - first = read(directory / 'metrics-window-start.json')[0]['metrics']['node_log_progress'] - last = read(directory / 'metrics-window-end.json')[0]['metrics']['node_log_progress'] + # A read-only window need not append. It still cannot hide epoch changes, + # reset counters or an invalid frontier. Positive offered writes always + # require advancement, even when every write failed before receiving proof. + valid = all(type(frontier.get('log_epoch')) is int and frontier['log_epoch'] > 0 + and all(type(frontier.get(key)) is int and frontier[key] >= 0 + for key in ('issued_through', 'tiered_through', 'follower_proven_through')) + and frontier['tiered_through'] <= frontier['issued_through'] + and frontier['follower_proven_through'] <= frontier['issued_through'] + for frontier in frontiers) + if valid: + valid = len({frontier['log_epoch'] for frontier in frontiers}) == 1 and all( + after[key] >= before[key] + for before, after in zip(frontiers, frontiers[1:]) + for key in ('issued_through', 'tiered_through', 'follower_proven_through')) + first, last = frontiers[0], frontiers[-1] proven_start, proven_end = first.get('follower_proven_through'), last.get('follower_proven_through') - advancing = (isinstance(proven_start, int) and isinstance(proven_end, int) + advancing = (type(proven_start) is int and type(proven_end) is int and proven_end > proven_start) - passed = passed and advancing + passed = healthy and valid and (advancing or not writes_offered) return {'available': True, 'pass': passed, 'frontiers': frontiers, 'follower_proof_advanced': advancing, - 'reason': None if passed else 'follower shipping inactive, fenced, rotating or proof did not advance'} + 'follower_proof_required': writes_offered, 'frontier_valid': valid, + 'reason': None if passed else 'follower shipping inactive, fenced, rotating, invalid or required proof did not advance'} def case_report(directory): @@ -310,7 +327,7 @@ def case_report(directory): metrics = {'available': False, 'error': str(error)} point_failures.append('metric evidence invalid') writes = data['writes'] - fleet_mode = fleet_mode_report(path.parent) if case['system'] == 'cellule' and case['durability'] == 'fleet' else {'available': False, 'pass': False, 'reason': 'not a Cellule Fleet window'} + fleet_mode = fleet_mode_report(path.parent, writes_offered=writes['planned_offers'] > 0) if case['system'] == 'cellule' and case['durability'] == 'fleet' else {'available': False, 'pass': False, 'reason': 'not a Cellule Fleet window'} if case['system'] == 'cellule' and case['durability'] == 'fleet' and not fleet_mode['pass']: point_failures.append('active follower durability unverified') stability = stability_report(path.parent) if case['system'] == 'cellule' else {'available': False, 'pass': False, 'reason': 'celld publication age not exposed by this fixture'} diff --git a/scripts/tests/test_perf_report.py b/scripts/tests/test_perf_report.py index 4daea191..6371675f 100644 --- a/scripts/tests/test_perf_report.py +++ b/scripts/tests/test_perf_report.py @@ -32,7 +32,8 @@ def test_inactive_fleet_cannot_qualify_with_zero_publication_debt(self): def test_active_fleet_requires_healthy_frontiers_throughout_the_window(self): with tempfile.TemporaryDirectory() as temporary: directory = Path(temporary) - healthy = {'fleet_active': True, 'fenced': False, 'rotating': False} + healthy = {'fleet_active': True, 'fenced': False, 'rotating': False, + 'log_epoch': 1, 'issued_through': 2, 'tiered_through': 0} paths = [directory / f'metrics-window-{label}.json' for label in ('start', 'minute-1', 'end')] for index, path in enumerate(paths): @@ -49,6 +50,33 @@ def test_active_fleet_requires_healthy_frontiers_throughout_the_window(self): paths[1].write_text(json.dumps([{'metrics': {'node_log_progress': bad}}])) self.assertFalse(REPORT.fleet_mode_report(directory)['pass']) + def test_read_only_fleet_keeps_the_same_verified_frontier_without_new_appends(self): + with tempfile.TemporaryDirectory() as temporary: + directory = Path(temporary) + healthy = {'fleet_active': True, 'fenced': False, 'rotating': False, + 'log_epoch': 1, 'issued_through': 1000, 'tiered_through': 1000, + 'follower_proven_through': 1000} + paths = [directory / f'metrics-window-{label}.json' + for label in ('start', 'minute-1', 'end')] + for path in paths: + path.write_text(json.dumps([{'metrics': {'node_log_progress': healthy}}])) + self.assertFalse(REPORT.fleet_mode_report(directory)['pass']) + idle = REPORT.fleet_mode_report(directory, writes_offered=False) + self.assertTrue(idle['pass']) + self.assertFalse(idle['follower_proof_required']) + self.assertFalse(idle['follower_proof_advanced']) + # An idle read window still cannot qualify after loss of Fleet, + # a reset/rotation or an impossible proof. No write gate changes. + for bad in (dict(healthy, fleet_active=False), dict(healthy, fenced=True), + dict(healthy, rotating=True), dict(healthy, log_epoch=2), + dict(healthy, follower_proven_through=999), + dict(healthy, follower_proven_through=1001), + dict(healthy, tiered_through=1001), + dict(healthy, follower_proven_through=True), + dict(healthy, error='lease lost')): + paths[1].write_text(json.dumps([{'metrics': {'node_log_progress': bad}}])) + self.assertFalse(REPORT.fleet_mode_report(directory, writes_offered=False)['pass']) + def test_missing_fleet_frontiers_cannot_qualify(self): with tempfile.TemporaryDirectory() as temporary: directory = Path(temporary) From 417eaacb0ee4718ce407e04dcd0e8450f3d8a867 Mon Sep 17 00:00:00 2001 From: forhappy Date: Wed, 7 Oct 2026 06:48:33 -0700 Subject: [PATCH 013/102] Record shared write qualification failures and proof obligations --- docs/bundle-coverage-proof.md | 52 ++- docs/write-performance-delivery.md | 563 +++++++++++------------------ 2 files changed, 248 insertions(+), 367 deletions(-) diff --git a/docs/bundle-coverage-proof.md b/docs/bundle-coverage-proof.md index 33ab59d7..43a03b41 100644 --- a/docs/bundle-coverage-proof.md +++ b/docs/bundle-coverage-proof.md @@ -29,30 +29,39 @@ coalesce more root work because a follower proof may respond earlier, but it still needs bounded retention and eventual object coverage; a longer root timer cannot release an unproved capture or hide its debt. -The corrected M2/M3 Docker candidate at `d351d886` completed three 300-second -windows at 100 offered writes/s. Scheduled p99 was 15.6 / 36.4 / 31.2 ms with +The latest M2/M3 Docker candidate at `a3787a8d` completed three 300-second +windows at 100 offered writes/s. Scheduled p99 was 19.8 / 23.1 / 46.9 ms with zero errors/drops/unissued offers; every run's 34,001 seed/warmup/window ACKs passed warm and cold GET/retry audits. Fleet remained active and follower -proofs advanced. All three debt trends failed and total PUT cost was -5.330–5.484 per command. These are diagnostic points, not a qualified capacity +proofs advanced. Two debt trends failed and total PUT cost was +5.229–5.433 per command. These are diagnostic points, not a qualified capacity or parity claim. The first window supplies the detailed observations below. | Window observation | Value | Remaining work | | --- | ---: | --- | -| All provider PUT successes/command | 5.4842 | Amortize authority selection, not only payload upload | -| Cell-authority PUT successes/command | 2.2521 | Root lineage and Cell selection remain per-Cell work | -| Immutable PUT successes/command | 2.2499 | Includes payloads, roots and maintenance | -| Node-authority PUT successes/command | 0.9822 | Coverage selection is almost per-command at this arrival rate | +| All provider PUT successes/command | 5.4330 | Amortize authority selection, not only payload upload | +| Cell-authority PUT successes/command | 2.2507 | Root lineage and Cell selection remain per-Cell work | +| Immutable PUT successes/command | 2.2422 | Includes payloads, roots and maintenance | +| Node-authority PUT successes/command | 0.9401 | Coverage selection is almost per-command at this arrival rate | | Fresh owner + receiver enrollment GETs/command | 0.0138 | Grant budget passes in this single window; paired qualification remains required | -| Cells/shared cohort | 1.0023 | At this sparse arrival rate the bounded coordinator mostly flushes singletons | +| Singleton Cell submissions | 98.64% | At this sparse arrival rate the coordinator uses canonical native packs | +| Cells/multi-row shared cohort | 2.7020 | Only 151 multi-row cohorts; this excludes singletons | | Commands/selected Cell root | 1.0000 | Delaying a Cell to collect many commands cannot meet the sparse latency target | -The strict three-minute trend gate failed: unpublished debt rose by -98.53 bytes/s and oldest debt by 0.152 ms/s in the fitted tail. These slopes +The first window's strict three-minute trend gate failed: unpublished debt rose by +34.80 bytes/s and oldest debt by 0.0817 ms/s in the fitted tail. These slopes must remain failures in the evidence; one run cannot distinguish sustained growth from sampling variation. Increasing cohort delay without a measured latency/debt benefit is not a remedy. +At the 15K Fleet target, the same candidate accumulated 70.64 MB of unpublished +log debt and its object frontier lagged by 50,548 sequences. It completed only +295.263 commands/s, failed delivery and had 432 warm-audit HTTP 503s. The +[delivery report](write-performance-delivery.md) preserves this failure. +Follower acknowledgments keep the immediate proof path short only while the +publication/retention path can make progress. M4 must reduce the complete cost +and bound unmaterialized locators, with headroom for leases and drain. + ## Authority and transfer Use the existing canonical node record's mutable log authority for both the @@ -162,6 +171,15 @@ retain three root/lineage/Cell-selection PUTs, the remaining M4 budget requires checkpoint, before compaction, catalog changes or retries. If manifests need a separate PUT, three requests per 64 commands leave even less checkpoint budget. +The 160-command value is an optimistic metadata lower bound. A checkpoint that +rewrites bundle dependencies into one new independent data object has at least +four PUTs, requiring at least 214 commands per checkpoint under the same +two-PUT/64 assumption, before indexes, multipart requests or maintenance. A +root that keeps shared payload references may shorten the manifest scan, but it +cannot release those payload objects. Account for lookup checkpoints and actual +data reclamation separately; independent checkpoints cannot be treated as free +uploads. + At the uniform 1,000-Cell bucket target of 2,000 commands/s, 160 commands per Cell span approximately 80 seconds; at 15,000 Fleet commands/s they span about 10.7 seconds. These are inferred cost constraints, not measured performance or @@ -219,6 +237,14 @@ precede object selection. Closing freezes issuance and drains exactly that old range before terminal closure and reconstruction. Its positive profile also checks eventual closing under weakly fair successful upload/selection. Three negative profiles require counterexamples for selected-only transfer, late -issuance and premature follower retirement. Verification is pending separately -from the state counts above. Exact captures/checkpoints and the Rust join/close +issuance and premature follower retirement. At `954cacf`, CI passed all 801 +distinct positive states, the fair closing property and the three named +counterexamples. The selected-only trace is +`Issue → Replicate → FleetAck → BeginClose → FinishClose → Transfer`: the +new writer's reconstructed endpoint omits the acknowledged command. +Exact captures/checkpoints and the Rust join/close barrier remain production obligations, not proofs supplied by this abstraction. +The [three-model CI run](https://github.com/crabbuild/cellule/actions/runs/37620661618) +preserves all eleven required unsafe counterexamples across the independent +models. Their state counts cannot be combined into a proof of one production +protocol. diff --git a/docs/write-performance-delivery.md b/docs/write-performance-delivery.md index cf8bbe40..f37afff3 100644 --- a/docs/write-performance-delivery.md +++ b/docs/write-performance-delivery.md @@ -38,10 +38,10 @@ collection paths. There is no legacy decoding or automatic migration. | --- | --- | --- | | M0 | Measurement and comparison harness delivered | Three A/A capacity pairs unverified; storage API totals reconcile, but SDK-internal HTTP retries need provider telemetry | | M1 | Packs, inline leaves and bounded compaction spooling delivered | Ordinary append meets four PUTs; composed compaction needs five. Paired cold/sparse-read guardrail unverified | -| M2 | Bounded file-backed shared publication coordinator implemented; exact scope, restore, cancellation, minimum-budget and dormant-sibling retention checks added | Three corrected active-Fleet windows cost 5.330–5.484 PUTs/command and all fail the debt trend. Three per-Cell authority PUTs remain; M4 is required | -| M3 | Signed 512-sequence/five-second grants, bounded local registry, lifecycle gate and signed HTTP fixture implemented | Full isolated checks and native lifecycle suite pass; three active-Fleet windows cost 0.0138–0.0141 enrollment GETs/command. Target-rate qualification remains unverified | +| M2 | Bounded file-backed shared publication coordinator implemented; exact scope, restore, cancellation, minimum-budget and dormant-sibling retention checks added | Latest three active-Fleet windows cost 5.229–5.433 PUTs/command; two fail the debt trend. Three per-Cell authority PUTs remain; M4 is required | +| M3 | Signed 512-sequence/five-second grants, bounded local registry, lifecycle gate and signed HTTP fixture implemented | Full isolated checks and native lifecycle suite pass; latest three active-Fleet windows cost 0.0138 enrollment GETs/command. The 15K target diagnostic fails delivery and warm audit | | M4 | Binding/selector and delayed-materialization models plus [authority decision](bundle-coverage-proof.md) delivered | Production bundle proof, atomic transfer/recovery/collection and bundle ACKs not implemented | -| M5 | Three paired low-rate Fleet repetitions and earlier read/target stress with exact ACK audits delivered | Publication stability fails; qualified capacity, read/failure/overload matrix and absolute/relative parity remain unverified | +| M5 | Three paired low-rate Fleet repetitions and target diagnostics with exact ACK audits delivered | Publication stability and target delivery fail; qualified capacity, read/failure/overload matrix and absolute/relative parity remain unverified | ## Shared publication checkpoint @@ -113,7 +113,7 @@ isolated verification passed 1,862 workspace tests (38 ignored), 58 local LTX tests, all contributor checks and the complete 382-test serial native lifecycle suite. The release binary is pinned to `a3787a8d`; all 1,248 Rust/Cargo source hashes match the verified snapshot. Three matched Docker repetitions and fresh -routing CI are tracked separately. The `d351d886` results below +routing CI are recorded below. The `d351d886` results further below must not be attributed to this subsequent change. Fresh routing CI for `75ae439d` passed all four paired command comparisons in @@ -125,12 +125,155 @@ Their CI latency limit is two times baseline, not the proposal's 1.2 read limit: object-only forwarded/local concurrency-one query p99 ratios were 1.614 / 1.241. Passing routing CI therefore does not pass the M5 read guardrail. -The same head's workspace CI failed one combined-maintenance lifecycle test with -a nested fleet-action-journal conflict (381/382 passed); a fresh rerun is -pending. The isolated complete native suite passed all 382 on the same production -source. Both results are retained; the CI conflict's cause is not established. +The same head's workspace CI initially failed one combined-maintenance lifecycle +test with a nested fleet-action-journal conflict (381/382 passed). The complete +unchanged-source rerun passed, as did the isolated 382-test native suite. Both +CI results are retained; a passing rerun does not establish the conflict's cause. +Fresh [Rust CI at `954cacf`](https://github.com/crabbuild/cellule/actions/runs/37620661541) +also passes all 382 native lifecycle tests. That revision changes reporting and +models, not the frozen production Rust/Cargo source. -## Corrected shared/grant measurement +## Latest singleton/shared/grant measurement + +Release candidate `a3787a8d`, current main `831877cf` and celld `f2bf6486` +completed three matched Fleet repetitions at 100 offered writes/s. Each used +300 seconds plus 30 seconds warmup, 1,000 uniform Cells, 96-byte SQL-ledger +values and 128 clients on the shared Docker VM. System order alternated; +clients, auditors, images and workload contracts match. Neither Cellule arm +uses a measurement overlay. + +| System | Window commands/s, repetitions 1 / 2 / 3 | Scheduled p99, ms, repetitions 1 / 2 / 3 | Point delivery/latency result | +| --- | --- | --- | --- | +| Main `831877cf` | 100.000 / 99.977 / 100.000 | 123.9 / 140.2 / 158.6 | All exceed 50 ms | +| Candidate `a3787a8d` | 100.000 / 100.000 / 100.000 | 19.8 / 23.1 / 46.9 | All pass; two publication-debt trends fail | +| celld `f2bf6486` | 100.000 / 100.000 / 100.000 | 15.9 / 15.6 / 18.2 | All pass; publication-debt age unavailable | + +All nine cases checked every one of 34,001 ACKs with warm GET/retry and cold +GET/retry, with zero audit failures. All original owner/follower containers +were destroyed after drain and before cold restore. Candidate Fleet remained +active, unfenced and non-rotating with advancing valid follower frontiers in +all three windows; its drain took 5.41 / 8.29 / 15.50 seconds. + +Candidate total PUTs/command were 5.4330 / 5.3979 / 5.2295, versus main's +5.4189 / 5.4156 / 5.3684. Summed fresh owner/receiver enrollment GETs/command +were 0.0138 in each candidate window. Candidate debt slopes were ++34.80 / +18.58 / −239.85 bytes/s and oldest-age slopes were ++0.0817 / +0.0200 / −1.2583 ms/s. The first two remain failures; the third +passes. Main fails all three debt trends. This does not establish stable +capacity or a repeatable publication-cost reduction. + +The candidate used 29,592 / 29,456 / 27,950 singleton publications out of +30,000 Cell submissions per window. Multi-row cohorts accounted for +408 / 544 / 2,050 submissions; no pressure or large-input fallback occurred. +Near-singleton arrival patterns explain why payload sharing has little effect +on the per-Cell authority floor. The singleton optimization removes extra local +work while preserving the canonical native pack and exact root. + +In the first window, mean capture was 0.550 ms, Fleet proof 3.690 ms, Fleet +response 5.582 ms and publication 22.676 ms. These phases overlap and cannot +be added. The worker round trip was 1.522 ms, including queueing. Mean shared +upload was 43.744 ms over only 151 multi-row cohorts; it must not be compared +with earlier histograms that mixed singleton and multi-row uploads. Sampled +Docker CPU averaged approximately 0.51 owner cores and 1.63 of RustFS's two +cores. This identifies provider occupancy at this point, not CPU capacity or +a sustainable-rate ceiling. + +## Target-load failures + +One matched 300-second Fleet target pair offered 15,000 writes/s after 30 +seconds warmup. All three arms fail delivery qualification: + +| System | Window commands/s | Scheduled all-attempt p99, ms | Request errors / dropped offers | ACK audit outcome | +| --- | ---: | ---: | --- | --- | +| Main `831877cf` | 507.040 | 1,391.8 | 254,842 / 4,092,793 | All 198,356 ACKs pass warm/cold GET and retry; 15.04-s drain | +| Candidate `a3787a8d` | 295.263 | 143.1 | 3,667,746 / 743,667 | 432 warm 503s; cold not reached | +| celld `f2bf6486` | 1,008.373 | 117.8 | 2,714,671 / 1,482,817 | Owner self-fences; all warm requests fail transport; cold not reached | + +The candidate completes fewer commands than main at this overloaded point; +its lower all-attempt p99 includes fast failures and cannot establish a latency +or capacity win. Main coalesces 4.856 materialized commands/selected root, +versus the candidate's 2.080. These observations do not isolate a causal change, +but they require preserving coalescing and retained-memory headroom when making +the follower proof path faster. Main also fails the debt-age trend despite +remaining in active Fleet throughout its window. + +The frozen candidate's 15,000 offered Fleet writes/s diagnostic completed +295.263 commands/s inside its 300-second window. It had 3,667,746 request +errors, 743,667 dropped offers and scheduled all-attempt p99 of 143.1 ms. +Its 104,442 actual seed/warmup/window/trailing ACKs reconcile with the journals, +but warm audit returned 432 HTTP 503 errors. Cold restore was not reached; +the case is failed, not a data-loss or durability success claim. + +Follower proofs advanced while sampled object coverage fell 50,548 sequences +behind. Tail debt grew by 361,853 bytes/s and oldest age by 837.56 ms/s. +Window PUT cost of 1.750/command excludes its substantial unpublished suffix +and trailing work; it cannot be credited as an improvement. Dirty admission +averaged 348 ms and publication 3,968 ms, while capture averaged 1.282 ms. +These are overlapping waits/populations, not additive CPU service times. +The final sample had 997 active Cells and 54.79 MiB of the 64-MiB retained +budget charged. The audit records status but not the underlying source error, +so pressure eviction, pending logical proof and unfinished accepted work remain +diagnostic possibilities rather than an established cause. + +The two followers' window mean durable append times were 16.61 / 17.21 ms +per append operation; native worker time was 16.30 / 16.87 ms and worker queue +wait 0.158 / 0.185 ms. Their data-sync means were only 0.0010 / 0.0021 ms +on tmpfs. These populations are batched append operations, not commands. +This profile does not identify filesystem sync as the dominant wait and cannot +answer physical-storage fsync versus RocksDB performance. Preserve the native +barrier; profile the remaining worker work in a separate diagnostic run. +The native append path calls `prune_covered`, which scans and hashes current +open-chunk records before appending and can rewrite/reconcile after coverage +advances. This is a concrete source of additional work, not a measured fraction +of the 16–17-ms worker time. Measure scan bytes, record count and rewrite time +before changing it; current-disk corruption and uncovered-witness checks cannot +be replaced by an unchecked cached watermark. + +The matched celld Fleet target completed 1,008.373 commands/s before delivery +failure. Its owner evicted a gray follower, temporarily fell back to bucket, +then self-fenced after an ambiguous lease renewal and exited with code 3. +All 440,676 warm audit requests failed transport; cold restore was not reached. +Neither target point qualifies capacity or the reported laptop's bounded-KV +result. Failed cases and exact binaries remain retained outside Git. + +The matched bucket target offered 2,000 writes/s under the same resources: + +| System | Window commands/s | Scheduled all-attempt p99, ms | Request errors / dropped offers | ACK audit outcome | +| --- | ---: | ---: | --- | --- | +| Main `831877cf` | 182.913 | 7,900.9 | 0 / 544,870 | All 61,988 ACKs pass warm/cold GET and retry; 6.01-s drain | +| Candidate `a3787a8d` | 172.283 | 6,632.2 | 0 / 548,059 | All 59,526 ACKs pass warm/cold GET and retry; 7.43-s drain | +| celld `f2bf6486` | 756.387 | 813.5 | 0 / 372,827 | All 260,506 warm checks pass; cold has 137 HTTP 500 errors | + +All measured offers were generated, but every arm dropped requests in both +warmup and measurement. Candidate PUT cost was 3.823/command versus main's +4.220, with 1.026 versus 1.045 materialized commands/root. Candidate publication +age grew 17.37 ms/s and retained-capture debt grew 464.38 bytes/s; main also +fails age growth. The lower window PUT ratio does not qualify M2 or stable +capacity. Celld's provider remains healthy through cold and final snapshots; +the 137 cold errors' source is unresolved. They cannot be attributed to the +historical provider OOM or labeled acknowledged-state loss without further +evidence. All six target cases remain in the matched comparisons. + +## Qualification decision + +This delivers the quantified gap report permitted by M5. It does not qualify +rollout or complete M0–M5. Each target has only one matched pair; three +repetitions at 100/s are diagnostic points rather than a capacity search. + +| Gate | Current result | +| --- | --- | +| Contributor checks, native lifecycle and bounded protocol models | Pass within their documented scope | +| Latest three Fleet 100/s delivery/latency and every warm/cold ACK retry | Pass; candidate remains slower than celld in each p99 comparison | +| Three stable Fleet windows and M2's 0.25 PUT/command budget | Fail: two candidate debt trends grow; cost 5.229–5.433 | +| 15K Fleet and 2K bucket absolute targets | Fail in every arm; candidate also completes fewer overloaded commands than main | +| M3's 0.05 fresh enrollment GET budget | Within budget at the three 100/s points; full qualification remains unverified | +| M4's production proof, atomic fault matrix and 0.05 total PUT budget | Unimplemented; abstract models do not enable ACKs or GC | +| Read-only/mixed capacity and 1% hot-Cell guardrails | Unverified in Docker; two native routing p99 ratios exceed 1.2 | +| A/A capacity variance and relative parity | Unverified; overloaded completions cannot supply the reference | +| Qualified overload, safe refusal before SQL and immediate recovery | Unverified; the historical step-down failed latency | +| Bounded KV and persistent-device failure profile | Unverified; SQL-ledger/tmpfs results do not establish these | + +## Earlier corrected shared/grant measurement Candidate `d351d886` and main `831877cf` use separate source-content-isolated release builds, with no measurement overlay. Celld is pinned to v0.6.1 @@ -138,6 +281,12 @@ release builds, with no measurement overlay. Celld is pinned to v0.6.1 the arms. The current profile is the SQL ledger workload below on one shared 8-CPU/16-GiB VM, with tmpfs node state and fresh RustFS volumes. +Healthy celld shipping waits for every selected follower; the fixture checks +initial enrollment of both followers. Celld can reconfigure down to one follower, +while this Cellule configuration retains two required members. Fault/availability +qualification must align that policy separately. Celld's publication-debt age +and structured window membership are unavailable in the current fixture. + Three paired repetitions offered 100 writes/s for 300 seconds after 30 seconds warmup, alternating system order. All original node containers were removed after drain and before each bucket-only cold audit. These are matched diagnostic @@ -161,31 +310,10 @@ Candidate PUTs/command were 5.4842 / 5.3301 / 5.3969 and summed fresh enrollment GETs/command were 0.0138 / 0.0141 / 0.0138. Candidate tail-debt slopes were +98.53 / +1,588.95 / +12.54 bytes/s; all remain failures. Celld debt age is not exposed by this fixture, so its stability is unavailable rather than passing. -The first candidate window supplies the phase and cohort details below. - -| Candidate window/audit | Measured value | -| --- | ---: | -| Successful commands/s inside window | 100.000 | -| Scheduled p50 / p95 / p99, ms | 5.5 / 7.9 / 15.6 | -| Errors / dropped / unissued offers | 0 / 0 / 0 | -| Warm and cold GET plus exact retry checks | 34,001 each; zero failures | -| Original fleet drain | 5.305 s | -| All provider PUT successes/command | 5.4842 | -| All GET attempts/command, including ranges | 2.4048 | -| Fresh enrollment GETs/command, owner + receivers | 0.0138 | -| Shared cohorts / Cell submissions | 29,931 / 30,000 | -| Shared pressure / large fallbacks | 0 / 0 | -| Logical commands/selected Cell root | 1.0000 | -| Final debt / oldest-age slope | +98.53 bytes/s / +0.152 ms/s | - -Fleet stayed active, unfenced and non-rotating throughout the sampled window; -its follower proof frontier advanced in every repetition. Delivery, latency and -all-ACK audits pass at these points. The strict publication stability gate fails, -and PUT cost is -well above M2's 0.25 budget. Near-singleton cohorts show why shared payloads -alone cannot amortize sparse per-Cell authority work. The -[bundle decision](bundle-coverage-proof.md) describes the remaining atomic -authority/recovery changes and cost accounting. +Fleet stayed active with advancing follower proofs in all three repetitions. +The publication stability and 0.25-PUT M2 budget still fail. These earlier +histograms include singleton shared uploads; their timings cannot be compared +with the latest multi-row-only histogram. Reports require Fleet activity and proof advancement independently of debt. An inactive fallback run cannot qualify Fleet even if it has zero enrollment @@ -193,323 +321,49 @@ GETs or no remaining follower debt. Window cost subtracts cumulative counters and raw histogram buckets; decreasing summary percentiles/means are not monotonic counters. Reporter source hashes accompany reevaluated evidence. -## Historical packed-implementation evidence - -The fixture is **SQL application parity**, not the user's bounded KV workload: -1,000 Cells, 96-byte values, INSERT plus in-command SELECT, and a two-hour -durable request/result ledger in both applications. Nodes have 8-CPU/16-GiB -ceilings and tmpfs state, RustFS has 2 CPUs/2 GiB, and the client has 4 CPUs/4 GiB -on one 8-CPU/16-GiB Linux VM. Summed CPU ceilings exceed VM capacity. This -profile qualifies neither device persistence nor independent-node isolation. - -The earlier baseline is `18eff0f7af47fac09b993157bb444e582072d7cf`, which merged PR 65. -Its production source matches the audited `397f500a` foundation; three additional -main files are design documents. The baseline explicitly overlays measurement -hooks. Celld is v0.6.1, `f2bf648663a610eefde71f3547ad61e9b896b1f0`, using the -pinned container digest. Both framework arms use byte-identical clients/auditors. - -Each arm offered 100 Fleet writes/s for 300 seconds after 30 seconds warmup, -serially with a fresh provider volume. These are individual points, not a -capacity search or three-repetition qualification. - -| Window or audit | Latest main + telemetry | Packed candidate | celld | -| --- | ---: | ---: | ---: | -| Successful commands/s inside window | 99.993 | 99.990 | 100.000 | -| Scheduled p50 / p95 / p99, ms | 66.8 / 265.3 / 868.1 | 24.0 / 200.1 / 386.5 | 5.6 / 11.8 / 44.3 | -| Errors / dropped offers | 0 / 0 | 0 / 0 | 0 / 0 | -| Commands checked by GET and exact retry, warm and cold | 34,001 each | 34,001 each | 34,001 each | -| Original fleet drain, seconds | 7.737 | 10.918 | 11.742 | -| All successful storage API PUTs/command | 6.688 | 4.992 | Not instrumented | -| All GET attempts/command, including ranges | 3.327 | 4.059 | Not instrumented | -| Fresh enrollment GETs/command, owner + receivers | 1.256 | 2.347 | Different authorization protocol | -| Logical commits per selected command root | 1.000 | 1.000 | Not instrumented | -| Final debt slope, bytes/s | +2,455.9 | +2,386.2 | Not instrumented | -| Final oldest-publication age slope, ms/s | +3.711 | +4.435 | Not instrumented | -| 50-ms delivery latency gate | Fail | Fail | Pass at this point | - -All original node containers were removed before bucket-only cold audits. This -checks recovery after successful drain, not owner loss before materialization. -The candidate lowered p99 by 55.5% and PUTs/command by 25.4% in this one pair, -but its p99 remains **8.72 times celld's**. Both Cellule arms fail the proposal's -publication stability gate. No sustainable-rate improvement is established. - -Compaction mean fell from 444.8 to 25.6 ms; publication mean from 1,288.7 to -489.4 ms; Fleet proof wait mean from 88.0 to 52.0 ms. Capture remained about -1.1 ms and tmpfs follower data sync about 0.0005–0.0007 ms. The faster candidate -also performed more compactions and enrollment GETs. Reduced batching is a -possible explanation for the enrollment increase; this run does not prove it. -The provider reached its two-CPU ceiling. Publication amplification and fresh -peer work remain priorities in this profile; device sync performance is untested. - -The measured one-command-per-root result cannot satisfy M2's authority-write -budget by sharing data alone. With three per-Cell selection PUTs, the floor is -three PUTs/command before shared data, node coverage or compaction. M3's measured -2.347 enrollment GETs/command is 46.9 times its 0.05 budget. Meeting those gates -requires the specified coalescing/proof/grant protocols and their failure tests. - -For the deterministic uniform schedule at 2,000 bucket writes/s, a Cell receives -one command every 500 ms. Waiting for its next command would already exceed the -200-ms bucket p99 target. The existing three-PUT per-Cell selection floor cannot -reach the proposed 0.05 PUT/command budget through cross-Cell data bundling alone. -This is a cost/latency bound under that schedule, not a measured capacity. M4 -needs a complete authority-pinned range proof before bucket responses can use -shared coverage; an uploaded bundle or live node epoch alone cannot supply it. - -### Fleet target stress and immediate step-down - -The rebuilt driver offered 15,000 writes/s for 300 seconds after 30 seconds -warmup. The candidate completed **476.80 writes/s inside the window**, with -3,224.5-ms scheduled p99, 61 request errors and 4,356,649 measured queue drops. -It generated every one of the 4.5 million scheduled offers. This is an overloaded -completion rate, not qualified capacity. All 190,175 successful acknowledgements -from setup, warmup, stress and both later phases passed warm and bucket-only cold -GET/exact-retry audits. Original fleet drain took 24.615 seconds. - -The window selected 28,468 command roots covering 145,103 logical commits: -5.097 commits/root. All provider PUT successes/acknowledged command fell to -1.000; fresh owner/receiver enrollment GETs/command were 0.083. These are -time-window ratios with maintenance included, not complete cohort accounting -of trailing work. Neither M2's 0.25 nor M3's 0.05 budget passed. - -Publication mean was 4,046 ms and dirty admission mean 5,409 ms. Worker round -trip averaged 9.28 ms and capture 0.65 ms. Follower worker calls averaged -47.9–48.4 ms, while their data-sync barriers averaged 0.0011–0.0013 ms on tmpfs. -The worker timing includes the native call; it does not identify its internal -CPU or filesystem costs. At window end, oldest unpublished work was 224.7 seconds -old and 76,435 node sequences awaited contiguous object coverage. The final -three-minute debt slope was +19,406.5 bytes/s and age slope +903.4 ms/s. - -Immediately after stress the harness offered 150 writes/s for 60 seconds, then -50 writes/s for 30 seconds, each without fresh warmup. Both had zero request -errors, queue drops or unissued offers, but scheduled p99 remained 471.6 and -495.0 ms. The 50/s phase failed latency recovery. The nominal reference of -100 writes/s was not qualified capacity; this exercises the phase/audit path, -not M5's required overload at 1.5 times a qualified reference. - -Celld's matching stress case failed through lease watchdog self-fencing: -both followers and the owner exited with code 3, without container OOM. -Its mixed healthy/failed window completed 627.14 writes/s with 7,071.1-ms p99, -435 request errors and 4,311,172 drops. The service was unavailable for its warm -audit of 337,803 acknowledgements; cold audit did not run. This establishes -availability and qualification failure, **not acknowledged-state loss or a -clean Fleet capacity comparison**. All failed journals, node logs and the -provider volume remain in the external artifacts. - -Latest main's retained attempt completed 121.84 writes/s with 3,683,013 request -errors and 780,436 queue drops, then zero successful step-down writes. Its -54,771 warm checks failed and drain exceeded 120 seconds. **This attempt is -excluded from framework performance attribution:** the shared provider ran out -of inodes by the end of the case, and the original harness had no inode samples. -The following bucket owner received S3 `InternalError: Disk full`; Linux reported -only five free inodes despite 44 GiB of free bytes. Main's window recorded 205 -transient PUT outcomes, including seven node-authority writes. The exact timing -and contribution of exhaustion are unknown. These failures cannot establish a -candidate availability advantage or a clean three-arm stress comparison. -The exclusion is retained in the machine-readable evidence, rather than deleting -the failed attempt. - -The first candidate and celld bucket attempts produced no throughput sample: -the former failed startup against the exhausted provider, and the latter could -not restart its control container. The dedicated Docker disk was expanded from -80 to 200 GiB without changing CPU/memory ceilings or tmpfs node state. The -runner now checks byte and inode headroom before, during and after each case; -missing or exhausted required observations fail qualification. One diagnostic -volume was archived with a SHA-256 manifest; other retained volumes remain. - -New artifact identities are `implementation-m1-bounded-audit` and -`implementation-main-bounded-audit`. Their SQL binary hashes remain identical -to the respective earlier builds; both use client `417f07b0424d` and auditor -`6595c24b0be2`. Fixture, runner and Docker host identities are recorded and must -match. Do not combine these results with the older client's paired point. - -### Fresh bucket target after storage reset - -With the drain-fixed candidate and monitored provider headroom, 2,000 offered -bucket writes/s for 300 seconds yielded 192.243 successful commands/s, zero -HTTP errors, 542,069 measured drops and 51,687 warmup drops. All 600,000 -measured offers were generated; scheduled p50/p95/p99 were 819.1/4,559.0/8,491.8 -ms. All 67,245 acknowledged commands passed warm and cold GET/exact-retry audits, -and original owner drain took 3.955 seconds. These overloaded completions do -not qualify sustainable capacity. - -All provider PUT successes/command were 4.241 and GET attempts/command 1.291; -the window materialized 1.038 commands per selected root. Worker round trip -averaged 1.539 ms and capture 0.535 ms, versus 460.5-ms publication and 648.1-ms -object-response wait. The 1,823 compaction observations averaged 5,707.4 ms. -These intervals overlap and must not be added into a latency breakdown. -Node-log debt was zero in bucket mode, but oldest-publication age had a positive -51.8-ms/s final trend. All 69 provider filesystem observations passed; minimum -free space was 166.3 GB and minimum free inodes 7,573,521. The earlier disk-full -failure is not an explanation for this target miss. - -Celld's matching window completed 905.077 writes/s with 1,075.3-ms scheduled -p99, three request errors, 328,218 measured drops and 24,986 warmup drops. All -600,000 offers were generated, and its warm audit passed all 307,794 acknowledged -commands and exact retries. At this overloaded point the candidate completed -21.24% of celld's rate, with 7.90 times its scheduled p99. These are point ratios, -not qualified-capacity ratios. Original owner drain took 9.761 seconds. During -cold audit the 2-GiB provider was OOM-killed at 05:27:57 UTC; the cold owner -then self-fenced after lease renewal failed. Only nine exact cold retries -completed. This is a provider availability failure: acknowledged-state loss is -unproven, and celld's cold durability is unverified in this case. Provider state -and the explicit cold-audit exclusion are retained in the report. - -Latest main's fresh matching window completed 104.710 writes/s, with zero HTTP -errors, 568,331 measured drops and 57,138 warmup drops. Scheduled p50/p95 were -1,367.0/7,131.3 ms; p99 exceeded the 10,000-ms histogram bound (954 overflow -samples), so no exact p99 or percentile ratio is reported. All 35,532 ACKs -passed warm and cold GET/exact-retry audits; original owner drain took 4.212 -seconds. All 70 filesystem observations passed, with at least 160.6 GB and -6,564,266 inodes free. - -The final framework build `19927d16` repeated the same offered point with the -same frozen runner, fixture, client, auditor and resources. It completed -163.033 writes/s, with zero HTTP errors, 550,834 measured drops and 54,079 warmup -drops; all 600,000 scheduled offers were generated. Scheduled p50/p95/p99 were -1,014.5/5,513.4/8,906.2 ms. All **56,088 ACKs** passed warm and cold GET/exact-retry -audits, and original owner drain took **9.656 seconds**. Cold startup took -32.710 seconds. Provider headroom passed all 69 filesystem observations, with -at least 158.2 GB and 6,136,282 inodes free. Oldest-publication age still grew -55.5 ms/s over the final three minute segments, so stability failed. - -| Fresh bucket target point | Latest main | Final packed candidate | celld | -| --- | ---: | ---: | ---: | -| Successful writes/s inside window | 104.710 | 163.033 | 905.077 | -| Scheduled p99, ms | >10,000 | 8,906.2 | 1,075.3 | -| HTTP errors / measured drops | 0 / 568,331 | 0 / 550,834 | 3 / 328,218 | -| Warm + cold ACKs checked by GET/exact retry | 35,532 each | 56,088 each | Warm 307,794; cold unavailable | -| Original owner drain, seconds | 4.212 | 9.656 | 9.761 | -| All successful storage API PUTs/command | 5.874 | 4.186 | Not instrumented | -| GET attempts including ranges/command | 1.592 | 1.218 | Not instrumented | -| Logical commands per selected root | 1.041 | 1.045 | Not instrumented | -| Delivery qualification | Fail | Fail | Fail | - -The final overloaded candidate completed 55.7% more writes than retained main, -with 28.7% fewer PUTs/command. It reached 18.01% of celld's measured rate and -8.28 times its scheduled p99. Logical value throughput was 15,651 bytes/s; -provider PUT bytes were 2.65 MB/s, including publication and coordination. -These are different numerators, not user payload versus wire-equivalent rates. -Its publication/object-response means were 536.7/766.4 ms; worker round trip -and capture averaged 1.852/0.644 ms. Compaction averaged 6,427.3 ms over 1,501 -observations. These overlapping intervals are not additive CPU service costs. - -The earlier candidate completed 83.6% more writes than main, with 27.8% fewer -PUTs/command and publication/object-response means of 460.5/648.1 ms. The final -candidate's completion rate was 15.2% lower than that sample. This is not an A/A -pair: the revision changed. Preserve both samples; three qualified A/A and -paired repetitions remain missing. Main's final publication-age trend was -negative; both candidate samples were positive. Neither the throughput ratios -nor successful audits establish sustainable capacity, stability improvement, -read guardrails or celld parity. The provider lifecycle checker now records -cold/final state as well as byte/inode headroom; an OOM or missing required -lifecycle observation fails future cases. - -The earlier immutable candidate is `a1be4caa`, artifact -`implementation-m1-drain-fixed`, SQL hash `7c774549a8f6`. The final candidate is -`19927d16`, artifact `implementation-m1-preserved-contract`, SQL hash -`0560f70c65fc`. The baseline artifact is -`implementation-main-drain-comparison`, SQL hash `85d30c0f93df`; client/auditor -hashes remain `417f07b0424d`/`6595c24b0be2`. Comparisons use the new identical -filesystem-monitoring runner `d70fad321f81`. Results from the former runner -remain separate. The final comparison is indexed by -`matched-verified-bucket-final-matrix.json` and -`matched-verified-bucket-final-report.json` outside the repository. - -The new lifecycle runner separately completed a five-second, one-write/s -diagnostic on the final binary: all 1,036 ACKs passed warm and cold GET/retry -audits, with all five required lifecycle observations healthy. The target -comparison deliberately retains its frozen runner; it does not acquire the -new runner's lifecycle evidence retroactively. A diagnostic is not capacity -qualification. - -### Read saturation evidence - -The same frozen clients offered 10,000 reads/s for five minutes after 30 seconds -warmup, without writes beyond setup. Every arm passed warm and cold GET/retry -audits of its 1,001 seed/contract mutations, with zero HTTP errors. Every arm -failed delivery qualification through dropped offers. These completion rates -are saturation observations, not qualified read capacities or M1's read guardrail. - -| Window | Latest main + telemetry | Packed candidate | celld | -| --- | ---: | ---: | ---: | -| Successful reads/s inside window | 9,893.81 | 9,934.98 | 9,321.13 | -| Scheduled p50 / p95 / p99, ms | 1.6 / 2.9 / 11.1 | 1.6 / 2.8 / 8.2 | 2.1 / 26.6 / 73.4 | -| Measured queue drops | 31,834 | 19,494 | 203,645 | -| Warmup queue drops | 14,240 | 902 | 12,138 | -| Unissued measured offers | 4 | 10 | 4 | -| Original fleet drain, seconds | 24.235 | 23.321 | 10.491 | - -Inspection and regression tests reproduced why the producer omitted those final -offers: its wall-clock stop could precede emission of an arrival already due -inside the window. The new producer emits the full scheduled cohort and keeps -lateness in the original arrival time. This fixes accounting; it cannot erase -the real queue drops in these historical runs. Three tests failed against the -old guard and passed after its removal. The updated client also accepts the -explicit zero-warmup overload/recovery phases used by the runner. - -The new auditor reads JSONL through bounded queues and checks every original -command, rather than retaining a multi-million-response vector. Collection -rejects duplicate IDs, missing successful journal records and partial input. -The runner verifies stream hashes before and after both audits and records its -loaded source and Docker host limits. New comparisons require matching fixture, -runner, host and client identities; older cases remain marked as lacking that -complete provenance. Rebuild both arms before comparing the revised driver; -historical and new clients are not interchangeable. - -| Frozen artifact | Run/build identity | SQL binary SHA-256 prefix | -| --- | --- | --- | -| Latest main + measurement overlay | `implementation-main-logical-metrics` | `85d30c0f93df` | -| Candidate | `implementation-m1-one-fetch` | `ff4c6ede25f4` | -| celld | v0.6.1 container digest pinned in `build.json` | Container digest | - -The client and auditor hashes are respectively `cc1d47522078` and -`35368139e4c5` for both builds. Full digests and every exported source hash live -in the retained manifests. All candidate production bytes match PR 66's -`a1c48fcd`; the final test expectation and documentation were edited after the -binary export. A Git base revision alone does not identify an overlaid build. - -The initial isolated all-feature workspace suite passed **1,839 tests** with **38 ignored** -environment-dependent tests. Clippy and API documentation passed with warnings -denied; format, boundaries, layout, Rust fences, links and SQL/peer contract -checks passed. The first compaction run exposed an old range-GET expectation; -the corrected test now requires zero range GETs and two complete pack GETs. -That failure remains in the external evidence. -The revised client/auditor passed 18 targeted Rust tests and warnings-denied -Clippy; local LTX without replica features passed 58 tests including its doctest. -The Python comparison, report, collector and provider checks passed 28 tests -with warnings treated as errors. These checks do not substitute for live overload, fault or -capacity qualification. - -The native fleet process suite exposed an empty-epoch shutdown loop after a -local owner fence. A regression test failed before the fix; the existing -fenced-owner evacuation test stalled at canonical shutdown. Empty coverage -queues now perform no new writer CAS, while pending tickets still reject -fencing and contiguous rotation/member/authority checks remain required. The -original process case passed in 1.30 seconds after the fix, and all 14 runtime -durability tests passed. The complete isolated fleet process suite then passed -all **382 tests** in 589.07 seconds. The full workspace suite, all-target/all-feature -check, warnings-denied Clippy/API documentation, format, architecture/layout, -document and SQL/peer gates passed again after the fix. Process tests are counted -separately from workspace tests. - -Final framework revision `19927d16` passed **1,843 workspace tests**, with -**38 ignored**, and all **382 fleet process tests** in 602.78 seconds. Its -all-target/all-feature check, Rust 1.99 warnings-denied Clippy, and API docs -passed; local LTX without replica features passed **58 tests** including its -doctest. All 1,240 Rust/Cargo source files in the isolated verification snapshot -match the checkout. The Linux SQL binary is `0560f70c65fc`; the client and -auditor remain `417f07b0424d` and `6595c24b0be2`. - -The recovery fault fixture now accepts only the original typed `Fenced` error -from the deliberately fenced source, retaining the `Draining` state and every -zero-resource-ledger assertion. Healthy receivers must still stop successfully. -A new deterministic case shuts down that source before receiver takeover, -checks that the selected authority record is unchanged, then verifies exact -reconstruction and the stored retry result. Production shutdown preserves -authority-release failures; closing local handles grants no successful release. -Three publication assertions also cover lease expiry, node fencing, and live -release. The failed contract-changing cleanup experiment and its test failures -remain outside Git as excluded evidence; that behavior was reverted. +## Historical results retained + +Earlier builds and clients remain separate evidence. They are not pooled with +`a3787a8d` or used as qualified capacity. The matched profile is the SQL ledger, +not bounded KV; tmpfs and the shared VM do not establish device persistence or +independent-node isolation. + +| Earlier profile | Main | Packed candidate | celld | Result | +| --- | --- | --- | --- | --- | +| Fleet, 100 offered/s | 99.993/s; p99 868.1 ms | 99.990/s; p99 386.5 ms | 100.000/s; p99 44.3 ms | All 34,001 ACKs/arm pass warm/cold GET and retry; both Cellule debt trends fail | +| Fleet, 15K offered/s | 121.84/s; provider-exhausted case excluded from attribution | 476.80/s; p99 3,224.5 ms | 627.14/s; p99 7,071.1 ms | All fail delivery; candidate audits 190,175 ACKs; celld self-fences before warm audit | +| Bucket, 2K offered/s, fresh provider | 104.710/s; p99 above 10,000-ms histogram bound | Final `19927d16`: 163.033/s; p99 8,906.2 ms | 905.077/s; p99 1,075.3 ms | All fail delivery; main/candidate cold audits pass, celld provider OOM during cold audit | +| Read-only, 10K offered/s, older client | 9,893.81/s; p99 11.1 ms | 9,934.98/s; p99 8.2 ms | 9,321.13/s; p99 73.4 ms | All drop offers; 4 / 10 / 4 final offers unissued; all 1,001 seed ACKs/arm pass warm/cold audit | + +The first Fleet point reduced candidate PUT cost from 6.688 to 4.992/command, +but candidate p99 remained 8.72 times celld's and its debt grew. The earlier +`a1be4caa` bucket candidate completed 192.243/s at p99 8,491.8 ms, versus final +`19927d16` at 163.033/s. The revision changed: this is not an A/A pair. Both +candidate bucket samples had positive publication-age trends. Final PUT cost +was 4.186/command and logical commits/selected root 1.045, preserving the +sparse per-Cell cost/latency gap. + +The old candidate's immediate Fleet step-down offered 150/s for 60 seconds +then 50/s for 30 seconds. It had zero delivery errors/drops but p99 remained +471.6 / 495.0 ms. The 100/s nominal reference was not qualified capacity; +this is a failed latency-recovery diagnostic, not the required overload gate. + +The provider-exhausted main attempt had only five free inodes despite 44 GiB +free bytes. Its contribution to failures is unknown; initial bucket attempts +produced no throughput sample. Later cases use fresh provider volumes plus +byte/inode and cold-lifecycle checks. Celld's later bucket case passed all +307,794 warm ACK/retry checks, then its 2-GiB provider was OOM-killed during cold +audit. This is availability failure with cold durability unverified, not proof +of acknowledged-state loss. All exclusions and failed journals remain retained. + +The producer's final-offer omission was reproduced and fixed; historical drops +and unissued requests remain failures. Bounded JSONL ACK collection now checks +every original retry, stream integrity and complete counts. The empty-epoch +shutdown loop was also reproduced and fixed without weakening pending-ticket +fencing. A cleanup experiment that weakened release contracts was reverted; +its failures remain excluded evidence. The final packed build passed 1,843 +workspace tests (38 ignored), 58 local LTX tests and 382 native lifecycle tests. +Current verification and source identities appear above. ## Reproduce and inspect @@ -520,9 +374,10 @@ and generated `report.json`. Content-based cache namespaces and persisted-root checks prevent stale codec reuse. Source contains reusable drivers and compact conclusions; caches, volumes, journals, metric windows, binaries and logs stay outside Git. Each retained provider volume has `store-data/volume.json`. -The external `delivery-evidence-index.json` hashes the final comparison, build, -case, runner, ACK stream, format smoke and verification manifest. The latter -pins the 1,240 Rust/Cargo sources and all final check logs. Failed and excluded +The external `singleton-r4-evidence-index.json` hashes the latest comparisons, +builds, case reports, loaded runners, ACK manifests and verification results. +`singleton-r4-qualification-status.json` records each passed, failed or unverified +gate. The production verification manifest pins all 1,248 Rust/Cargo sources. Failed and excluded attempts retain their own scope; they are not overwritten by passing reruns. `scripts/perf/compare.py` accepts an external JSON matrix containing `baseline`, From f7b20806c9e4e6a4bc8135b312bbc9141e3f558e Mon Sep 17 00:00:00 2001 From: forhappy Date: Wed, 7 Oct 2026 09:20:03 -0700 Subject: [PATCH 014/102] Add verified node bundle coverage and complete Cell drain --- crates/cellule-ltx/src/cell_layout.rs | 33 + crates/cellule-runtime/docs/storage.md | 4 + .../src/control/authority/mod.rs | 5 + crates/cellule-runtime/src/control/codec.rs | 35 +- crates/cellule-runtime/src/control/mod.rs | 61 +- .../src/node/advertisement/codec.rs | 45 +- .../src/node/advertisement/mod.rs | 20 + .../src/node/bundle/binding.rs | 156 +++++ .../src/node/bundle/closure.rs | 144 +++++ .../cellule-runtime/src/node/bundle/codec.rs | 220 +++++++ crates/cellule-runtime/src/node/bundle/mod.rs | 250 ++++++++ .../cellule-runtime/src/node/bundle/proof.rs | 243 ++++++++ .../src/node/bundle/selection.rs | 150 +++++ .../cellule-runtime/src/node/bundle/store.rs | 200 ++++++ .../src/node/bundle/tests/faults.rs | 136 ++++ .../src/node/bundle/tests/lifecycle.rs | 358 +++++++++++ .../src/node/bundle/tests/mod.rs | 586 ++++++++++++++++++ .../src/node/bundle/tests/ranges.rs | 313 ++++++++++ .../src/node/directory/advertisement.rs | 15 + .../src/node/directory/closure/mod.rs | 12 + .../cellule-runtime/src/node/directory/log.rs | 14 + .../cellule-runtime/src/node/directory/mod.rs | 31 +- .../src/node/directory/recovery.rs | 1 + .../src/node/durability/mod.rs | 23 +- crates/cellule-runtime/src/node/log/mod.rs | 264 +++++++- .../src/node/log_shipper/mod.rs | 33 +- .../src/node/log_shipper/tests.rs | 80 ++- crates/cellule-runtime/src/node/mod.rs | 1 + crates/cellule-runtime/src/publication/mod.rs | 31 + .../src/recovery/backup/mod.rs | 3 + .../src/recovery/retention/tests.rs | 1 + docs/bundle-coverage-implementation.md | 140 +++++ docs/bundle-coverage-proof.md | 9 +- docs/write-performance-delivery.md | 4 +- 34 files changed, 3588 insertions(+), 33 deletions(-) create mode 100644 crates/cellule-runtime/src/node/bundle/binding.rs create mode 100644 crates/cellule-runtime/src/node/bundle/closure.rs create mode 100644 crates/cellule-runtime/src/node/bundle/codec.rs create mode 100644 crates/cellule-runtime/src/node/bundle/mod.rs create mode 100644 crates/cellule-runtime/src/node/bundle/proof.rs create mode 100644 crates/cellule-runtime/src/node/bundle/selection.rs create mode 100644 crates/cellule-runtime/src/node/bundle/store.rs create mode 100644 crates/cellule-runtime/src/node/bundle/tests/faults.rs create mode 100644 crates/cellule-runtime/src/node/bundle/tests/lifecycle.rs create mode 100644 crates/cellule-runtime/src/node/bundle/tests/mod.rs create mode 100644 crates/cellule-runtime/src/node/bundle/tests/ranges.rs create mode 100644 docs/bundle-coverage-implementation.md diff --git a/crates/cellule-ltx/src/cell_layout.rs b/crates/cellule-ltx/src/cell_layout.rs index 18752bfc..e59076ee 100644 --- a/crates/cellule-ltx/src/cell_layout.rs +++ b/crates/cellule-ltx/src/cell_layout.rs @@ -68,6 +68,17 @@ impl CellStorageLayout { &self.application } + /// Binds application-scoped paths to another validated application under + /// the same canonical fleet prefix and transport. Node paths are shared. + #[must_use] + pub fn for_application(&self, application: [u8; 16]) -> Self { + Self { + store: self.store.clone(), + root: self.root.clone(), + application, + } + } + /// Returns the process-local identity used to isolate immutable read caches. #[must_use] pub fn immutable_cache_identity(&self) -> u64 { @@ -314,6 +325,22 @@ impl CellStorageLayout { )) } + /// Exact immutable node-wide coverage catalog and native range object. + #[must_use] + pub fn node_coverage_bundle_path( + &self, + session: &[u8; 16], + epoch: u64, + digest: &[u8; 32], + ) -> Path { + Path::from(format!( + "{}/cells/v1/node-logs/{}/{epoch}/coverage/v1/{}.cnb", + self.root, + encode_hex(session), + encode_hex(digest) + )) + } + fn application_path(&self, suffix: &str) -> Path { Path::from(format!( "{}/cells/v1/apps/{}/{}", @@ -365,6 +392,12 @@ mod tests { layout.node_directory_path().as_ref(), "tenant-root/cells/v1/nodes" ); + assert_eq!( + layout + .node_coverage_bundle_path(&[0xdd; 16], 7, &[0xef; 32]) + .as_ref(), + "tenant-root/cells/v1/node-logs/dddddddddddddddddddddddddddddddd/7/coverage/v1/efefefefefefefefefefefefefefefefefefefefefefefefefefefefefefefef.cnb" + ); assert_eq!( layout .node_log_bundle_path(&[0xdd; 16], 7, &[0xef; 32]) diff --git a/crates/cellule-runtime/docs/storage.md b/crates/cellule-runtime/docs/storage.md index 2f78e7b3..007680da 100644 --- a/crates/cellule-runtime/docs/storage.md +++ b/crates/cellule-runtime/docs/storage.md @@ -1,5 +1,9 @@ # Authority, storage, and recovery +The experimental [bundle coverage APIs](../../../docs/bundle-coverage-implementation.md) +separate selected native ranges from materialized roots. Bundle command ACKs +remain disabled pending actor, recovery, resource and collection integration. + One mutable authority record and one immutable object graph: identity, control, roots, pages, restore, compaction, backup, and retention. diff --git a/crates/cellule-runtime/src/control/authority/mod.rs b/crates/cellule-runtime/src/control/authority/mod.rs index 3b1afcc3..738bfdce 100644 --- a/crates/cellule-runtime/src/control/authority/mod.rs +++ b/crates/cellule-runtime/src/control/authority/mod.rs @@ -177,6 +177,11 @@ impl CellAuthority { transition: Transition, ) -> Result { observed.value.validate_transition(&next, transition)?; + if observed.value.bundle_binding.is_some() + && observed.value.bundle_binding != next.bundle_binding + { + crate::node::bundle::store::ensure_departure(&self.layout, &observed.value).await?; + } if observed.value.owner.is_some() && observed.value.owner != next.owner { // Preserve the original full control before the sole authority CAS // can erase its owner. A lost history reply cannot permit departure. diff --git a/crates/cellule-runtime/src/control/codec.rs b/crates/cellule-runtime/src/control/codec.rs index 7adf31fa..fc322905 100644 --- a/crates/cellule-runtime/src/control/codec.rs +++ b/crates/cellule-runtime/src/control/codec.rs @@ -2,7 +2,7 @@ use serde::{Deserialize, Serialize}; -use super::{Control, ControlState, Owner, RecoveryOverlayRef, RootRef}; +use super::{BundleBindingRef, Control, ControlState, Owner, RecoveryOverlayRef, RootRef}; use crate::identity::{CellId, Digest, IncarnationId, SessionId, decode_hex, encode_hex}; use crate::{Error, Result}; @@ -19,11 +19,21 @@ pub(super) struct RawControl { owner: Option, root: Option, recovery: Option, + #[serde(default, skip_serializing_if = "Option::is_none")] + bundle_binding: Option, code: String, schema: u32, next_due_ms: Option, } +#[derive(Serialize, Deserialize)] +#[serde(deny_unknown_fields)] +struct RawBundleBinding { + session: String, + epoch: String, + digest: String, +} + #[derive(Serialize, Deserialize)] #[serde(deny_unknown_fields)] pub(super) struct RawOwner { @@ -81,7 +91,11 @@ pub(super) struct RawRecoveryOverlay { impl From<&Control> for RawControl { fn from(control: &Control) -> Self { Self { - version: 1, + version: if control.bundle_binding.is_some() { + 2 + } else { + 1 + }, cell: encode_hex(control.cell.as_bytes()), incarnation: encode_hex(control.incarnation.as_bytes()), epoch: control.epoch.to_string(), @@ -113,6 +127,11 @@ impl From<&Control> for RawControl { final_commit_sequence: recovery.final_commit_sequence.to_string(), }), code: encode_hex(control.code.as_bytes()), + bundle_binding: control.bundle_binding.map(|binding| RawBundleBinding { + session: encode_hex(binding.session.as_bytes()), + epoch: binding.epoch.to_string(), + digest: encode_hex(binding.digest.as_bytes()), + }), schema: control.schema, next_due_ms: control.next_due_ms.map(|value| value.to_string()), } @@ -123,10 +142,20 @@ impl TryFrom for Control { type Error = Error; fn try_from(raw: RawControl) -> Result { - if raw.version != 1 { + if raw.version != if raw.bundle_binding.is_some() { 2 } else { 1 } { return Err(Error::Control("unsupported version")); } Ok(Self { + bundle_binding: raw + .bundle_binding + .map(|binding| { + Ok::<_, Error>(BundleBindingRef { + session: SessionId::from_bytes(decode_hex(&binding.session)?), + epoch: decimal_u64(&binding.epoch)?, + digest: Digest::from_bytes(decode_hex(&binding.digest)?), + }) + }) + .transpose()?, cell: CellId::from_bytes(decode_hex(&raw.cell)?), incarnation: IncarnationId::from_bytes(decode_hex(&raw.incarnation)?), epoch: decimal_u64(&raw.epoch)?, diff --git a/crates/cellule-runtime/src/control/mod.rs b/crates/cellule-runtime/src/control/mod.rs index 798a6bea..e6ffc2a5 100644 --- a/crates/cellule-runtime/src/control/mod.rs +++ b/crates/cellule-runtime/src/control/mod.rs @@ -90,6 +90,17 @@ pub struct Owner { pub endpoint: String, } +/// Exact node bundle lane pinned before it can cover this writer's commands. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub struct BundleBindingRef { + /// Original boot session whose canonical node record selects the bundle. + pub session: SessionId, + /// Immutable publication lane epoch. + pub epoch: u64, + /// Unique binding identity; a closed binding cannot be enrolled again. + pub digest: Digest, +} + /// Exact recovered follower tail pinned before a dead owner's Cell can move. #[derive(Clone, Debug, PartialEq, Eq)] pub struct RecoveryOverlayRef { @@ -170,6 +181,8 @@ pub struct Control { pub root: Option, /// Recovered overlay pinned before the Cell moved. pub recovery: Option, + /// Original bundle binding, retained until its complete issued range drains. + pub bundle_binding: Option, /// Application code digest the owner installed. pub code: Digest, /// Schema version the owner installed. @@ -181,6 +194,8 @@ pub struct Control { /// Named transition whose complete predicate must pass before an ETag update. #[derive(Clone, Copy, Debug, PartialEq, Eq)] pub enum Transition { + /// Pins a bundle binding without changing the writer or materialized root. + BindBundle, /// Extends the current owner's lease without changing the root. Renew, /// Installs a new owner for an existing root. @@ -229,6 +244,7 @@ impl Control { owner: Some(owner), root: None, recovery: None, + bundle_binding: None, code, schema, next_due_ms: None, @@ -281,6 +297,7 @@ impl Control { && previous.owner == self.owner && previous.root == self.root && previous.recovery == self.recovery + && previous.bundle_binding == self.bundle_binding && previous.code == self.code && previous.schema == self.schema && previous.next_due_ms == self.next_due_ms @@ -320,6 +337,7 @@ impl Control { /// Builds a recovering successor owned by a different enrolled session. pub fn takeover(&self, owner: Owner) -> Result { let mut next = self.clone(); + next.bundle_binding = None; next.epoch = next .epoch .checked_add(1) @@ -464,6 +482,7 @@ impl Control { /// Builds the sole valid successor that releases a drained Cell owner. pub(crate) fn release(&self) -> Result { let mut next = self.clone(); + next.bundle_binding = None; next.revision = next .revision .checked_add(1) @@ -490,7 +509,30 @@ impl Control { { return Err(Error::Control("successor revision or progress")); } + if !matches!( + transition, + Transition::BindBundle + | Transition::Release + | Transition::Takeover + | Transition::Tombstone + ) && self.bundle_binding != next.bundle_binding + { + return Err(Error::Control("successor changed bundle binding")); + } match transition { + Transition::BindBundle => { + let mut expected = self.clone(); + expected.revision = next.revision; + expected.progress = next.progress; + expected.bundle_binding = next.bundle_binding; + if self.state != ControlState::Serving + || self.bundle_binding.is_some() + || next.bundle_binding.is_none() + || expected != *next + { + return Err(Error::Control("invalid bundle binding transition")); + } + } Transition::Renew => { if self.epoch != next.epoch || self.state != next.state @@ -536,7 +578,8 @@ impl Control { } } Transition::Migrate => { - if !matches!(self.state, ControlState::Recovering | ControlState::Serving) + if self.bundle_binding.is_some() + || !matches!(self.state, ControlState::Recovering | ControlState::Serving) || next.state != ControlState::Serving || self.epoch != next.epoch || self.owner != next.owner @@ -552,6 +595,7 @@ impl Control { } Transition::Release => { if !matches!(self.state, ControlState::Recovering | ControlState::Serving) + || next.bundle_binding.is_some() || next.state != ControlState::Idle || next.owner.is_some() || next.root.is_none() @@ -610,6 +654,7 @@ impl Control { } Transition::Takeover => { if self.state == ControlState::Tombstoned + || next.bundle_binding.is_some() || next.state != ControlState::Recovering || next.owner.is_none() || self.owner == next.owner @@ -625,6 +670,7 @@ impl Control { } Transition::Tombstone => { if self.state == ControlState::Tombstoned + || next.bundle_binding.is_some() || next.state != ControlState::Tombstoned || next.owner.is_some() || self.epoch.checked_add(1) != Some(next.epoch) @@ -649,6 +695,19 @@ impl Control { if self.next_due_ms.is_some_and(|value| value < 0) { return Err(Error::Control("negative next due time")); } + if let Some(binding) = self.bundle_binding + && (binding.epoch == 0 + || binding.session.as_bytes().iter().all(|byte| *byte == 0) + || binding.digest.as_bytes().iter().all(|byte| *byte == 0) + || self.state != ControlState::Serving + || self.recovery.is_some() + || self + .owner + .as_ref() + .is_none_or(|owner| owner.session != binding.session)) + { + return Err(Error::Control("invalid bundle binding")); + } if let Some(owner) = &self.owner && (owner.endpoint.is_empty() || owner.endpoint.len() > 512 diff --git a/crates/cellule-runtime/src/node/advertisement/codec.rs b/crates/cellule-runtime/src/node/advertisement/codec.rs index 648f2d6b..fc8a424d 100644 --- a/crates/cellule-runtime/src/node/advertisement/codec.rs +++ b/crates/cellule-runtime/src/node/advertisement/codec.rs @@ -42,7 +42,7 @@ impl From<&NodeTombstone> for RawNodeTombstoneEnvelope { fn from(value: &NodeTombstone) -> Self { Self { tombstone: RawNodeTombstone { - version: 1, + version: if value.bundle.is_some() { 2 } else { 1 }, session: encode_hex(value.session.as_bytes()), node: encode_hex(value.node.as_bytes()), expires_at_ms: value.expires_at_ms.to_string(), @@ -53,6 +53,7 @@ impl From<&NodeTombstone> for RawNodeTombstoneEnvelope { claim_generation: value.claim_generation.to_string(), claim_expires_at_ms: value.claim_expires_at_ms.map(|value| value.to_string()), log: value.log.as_ref().map(encode_log), + bundle: value.bundle.map(RawNodeBundleHead::from), }, } } @@ -70,6 +71,8 @@ pub(in crate::node) struct RawNodeTombstone { pub(in crate::node) claim_generation: String, pub(in crate::node) claim_expires_at_ms: Option, pub(in crate::node) log: Option, + #[serde(default, skip_serializing_if = "Option::is_none")] + pub(in crate::node) bundle: Option, } impl From<&NodeAdvertisement> for RawUnsignedIdentity { @@ -104,6 +107,8 @@ pub(in crate::node) struct RawAdvertisement { pub(in crate::node) identity: RawIdentity, pub(in crate::node) lease: RawLease, pub(in crate::node) log: Option, + #[serde(default, skip_serializing_if = "Option::is_none")] + pub(in crate::node) bundle: Option, pub(in crate::node) capacity: RawCapacity, #[serde(default, skip_serializing_if = "Option::is_none")] pub(in crate::node) placement: Option, @@ -126,7 +131,7 @@ pub(in crate::node) struct RawIdentity { impl From<&NodeAdvertisement> for RawAdvertisement { fn from(value: &NodeAdvertisement) -> Self { Self { - version: 1, + version: if value.bundle.is_some() { 2 } else { 1 }, identity: RawIdentity { unsigned: RawUnsignedIdentity::from(value), signature: encode_hex(&value.signature), @@ -138,6 +143,7 @@ impl From<&NodeAdvertisement> for RawAdvertisement { expires_at_ms: value.expires_at_ms.to_string(), }, log: value.log.as_ref().map(encode_log), + bundle: value.bundle.map(RawNodeBundleHead::from), capacity: RawCapacity::from(value.capacity), placement: value.placement.map(|placement| RawPlacementCapacity { memory_capacity_bytes: placement.memory_capacity_bytes.to_string(), @@ -257,17 +263,50 @@ pub(in crate::node) struct RawNodeRecoveryClaim { pub(in crate::node) expires_at_ms: String, } +#[derive(Deserialize, Serialize)] +#[serde(deny_unknown_fields)] +pub(in crate::node) struct RawNodeBundleHead { + epoch: String, + digest: String, + selected_through: String, +} +impl From for RawNodeBundleHead { + fn from(head: crate::node::bundle::NodeBundleHead) -> Self { + Self { + epoch: head.epoch.to_string(), + digest: encode_hex(head.digest.as_bytes()), + selected_through: head.selected_through.to_string(), + } + } +} +impl TryFrom for crate::node::bundle::NodeBundleHead { + type Error = Error; + fn try_from(raw: RawNodeBundleHead) -> Result { + let head = Self { + epoch: canonical_u64(&raw.epoch)?, + digest: Digest::from_bytes(decode_hex(&raw.digest)?), + selected_through: canonical_u64(&raw.selected_through)?, + }; + head.validate()?; + Ok(head) + } +} + impl TryFrom for NodeAdvertisement { type Error = Error; fn try_from(value: RawAdvertisement) -> Result { - if value.version != 1 { + if value.version != if value.bundle.is_some() { 2 } else { 1 } { return Err(Error::Node("unsupported advertisement version")); } let raw = value.identity.unsigned; let session = SessionId::from_bytes(decode_hex(&raw.session)?); let node = NodeId::from_bytes(decode_hex(&raw.node)?); Ok(Self { + bundle: value + .bundle + .map(crate::node::bundle::NodeBundleHead::try_from) + .transpose()?, node, session, endpoint: raw.endpoint, diff --git a/crates/cellule-runtime/src/node/advertisement/mod.rs b/crates/cellule-runtime/src/node/advertisement/mod.rs index 81c7d400..1ff5d656 100644 --- a/crates/cellule-runtime/src/node/advertisement/mod.rs +++ b/crates/cellule-runtime/src/node/advertisement/mod.rs @@ -27,6 +27,7 @@ pub struct NodeAdvertisement { pub(super) failure_domain: NodeFailureDomain, pub(super) capacity: NodeCapacity, pub(super) log: Option, + pub(super) bundle: Option, pub(super) signature: [u8; 64], pub(super) placement_version: u32, pub(super) placement_signature: [u8; 64], @@ -75,6 +76,7 @@ impl NodeAdvertisement { failure_domain, capacity, log: None, + bundle: None, signature: [0; 64], placement_version: 0, placement_signature: [0; 64], @@ -285,6 +287,11 @@ impl NodeAdvertisement { self.log.as_ref() } + /// Immutable node bundle pointer; a sampled head is not a durability proof. + pub const fn bundle_head(&self) -> Option { + self.bundle + } + pub(super) fn encode(&self) -> Result> { self.validate_shape()?; self.verify_signature()?; @@ -327,6 +334,16 @@ impl NodeAdvertisement { } pub(super) fn validate_shape(&self) -> Result<()> { + if let Some(bundle) = self.bundle { + bundle.validate()?; + if self + .log + .as_ref() + .is_some_and(|log| log.epoch() != bundle.epoch()) + { + return Err(Error::Node("node bundle and native log epochs differ")); + } + } if self.node.as_bytes().iter().all(|byte| *byte == 0) || self.session.as_bytes().iter().all(|byte| *byte == 0) || self.fleet.as_bytes().iter().all(|byte| *byte == 0) @@ -448,6 +465,8 @@ impl NodeAdvertisement { let mut unsigned = RawAdvertisement::from(self); unsigned.placement_signature = None; unsigned.log = None; + unsigned.bundle = None; + unsigned.version = 1; // The directory assigns the monotonic heartbeat generation during a // CAS refresh; the signed lease timestamps/progress remain immutable // evidence while this server-owned counter is intentionally excluded. @@ -630,6 +649,7 @@ pub(super) fn validate_successor( } if !same_boot_identity(current, next) || current.log != next.log + || current.bundle != next.bundle || current.generation.checked_add(1) != Some(next.generation) || next.progress < current.progress || next.issued_at_ms <= current.issued_at_ms diff --git a/crates/cellule-runtime/src/node/bundle/binding.rs b/crates/cellule-runtime/src/node/bundle/binding.rs new file mode 100644 index 00000000..29461aac --- /dev/null +++ b/crates/cellule-runtime/src/node/bundle/binding.rs @@ -0,0 +1,156 @@ +//! Original Cell pin enrollment in the complete node catalog. +use super::*; +use crate::control::authority::{CellAuthority, VersionedControl}; +use crate::control::{ControlState, Transition}; +use crate::identity::ApplicationId; +use crate::node::{NodeDirectory, VersionedNodeAdvertisement}; + +impl NodeDirectory { + /// Creates an empty bundle lane before native issuance starts for this boot. + /// The caller must hold startup admission closed until initialization and + /// original Cell binding enrollment finish. + pub async fn initialize_bundle_lane( + &self, + observed: &VersionedNodeAdvertisement, + epoch: u64, + now_ms: i64, + ) -> Result { + if observed.advertisement().bundle_head().is_some() + || epoch == 0 + || observed + .advertisement() + .log() + .is_some_and(|log| log.active() || log.tiered_through() != 0) + { + return Err(Error::Node("bundle lane is already initialized or invalid")); + } + let catalog = Catalog { + session: observed.advertisement().session(), + epoch, + predecessor: None, + selected_through: 0, + bindings: Vec::new(), + }; + let prepared = self.upload_catalog(None, catalog, &[]).await?; + self.select_catalog(observed, &prepared, now_ms).await + } + + /// Pins one original writer in Cell authority before enrolling it in the + /// complete catalog. An ambiguous failure retains the pin and blocks departure. + pub async fn bind_bundle_cell( + &self, + observed: &VersionedNodeAdvertisement, + authority: &CellAuthority, + control: &VersionedControl, + now_ms: i64, + ) -> Result<(VersionedNodeAdvertisement, VersionedControl)> { + self.validate(&observed.advertisement, now_ms)?; + let head = observed + .advertisement + .bundle + .ok_or(Error::Node("bundle lane is absent"))?; + let mut catalog = load_catalog(&self.layout, observed.advertisement.session, head).await?; + let value = control.value(); + if value.state != ControlState::Serving + || value.recovery.is_some() + || value + .owner + .as_ref() + .is_none_or(|owner| owner.session != catalog.session) + { + return Err(Error::Fenced); + } + if authority.layout().node_path(catalog.session.as_bytes()) + != self.layout.node_path(catalog.session.as_bytes()) + || authority.layout().immutable_cache_identity() + != self.layout.immutable_cache_identity() + { + return Err(Error::Fenced); + } + let application = ApplicationId::from_bytes(*authority.layout().application_id()); + let pinned = if let Some(pin) = value.bundle_binding { + if pin.session != catalog.session || pin.epoch != catalog.epoch { + return Err(Error::Fenced); + } + control.clone() + } else { + let mut identity = value.encode()?; + identity.extend_from_slice(b"cellule.bundle-binding.v1\0"); + identity.extend_from_slice(application.as_bytes()); + identity.extend_from_slice(&catalog.epoch.to_le_bytes()); + let mut next = value.clone(); + next.revision = next + .revision + .checked_add(1) + .ok_or(Error::Control("revision overflow"))?; + next.progress = next + .progress + .checked_add(1) + .ok_or(Error::Control("progress overflow"))?; + next.bundle_binding = Some(BundleBindingRef { + session: catalog.session, + epoch: catalog.epoch, + digest: Digest::from_bytes(*blake3::hash(&identity).as_bytes()), + }); + match authority + .transition(control, next.clone(), Transition::BindBundle) + .await + { + Ok(pinned) => pinned, + Err(source) => match authority.load(value.cell).await { + Ok(Some(current)) if current.value() == &next => current, + _ => return Err(source), + }, + } + }; + let pin = pinned + .value() + .bundle_binding + .ok_or(Error::Node("Cell bundle pin missing"))?; + if let Some(binding) = catalog + .bindings + .iter() + .find(|binding| binding.control.bundle_binding == Some(pin)) + { + if binding.phase != BindingPhase::Open || binding.control != *pinned.value() { + return Err(Error::Fenced); + } + return Ok((observed.clone(), pinned)); + } + // Enrolling two writer bindings for one Cell would allow a departed + // epoch to rejoin under a new identity. Closed originals remain tombstones. + if catalog.bindings.iter().any(|binding| { + binding.application == application + && binding.control.cell == value.cell + && binding.phase != BindingPhase::Closed + }) { + return Err(Error::Fenced); + } + let base = pinned + .value() + .ltx_root() + .ok_or(Error::Node("bundle binding lacks published base"))?; + catalog.bindings.push(Binding { + application, + first_commit: base.commit_sequence, + control: pinned.value().clone(), + phase: BindingPhase::Open, + terminal: None, + selected_sequence: 0, + selected_commit: base.commit_sequence, + selected_position: base.position, + locators: Vec::new(), + }); + catalog.bindings.sort_unstable_by_key(|binding| { + binding + .control + .bundle_binding + .map(|pin| *pin.digest.as_bytes()) + }); + let prepared = self.upload_catalog(Some(head), catalog, &[]).await?; + Ok(( + self.select_catalog(observed, &prepared, now_ms).await?, + pinned, + )) + } +} diff --git a/crates/cellule-runtime/src/node/bundle/closure.rs b/crates/cellule-runtime/src/node/bundle/closure.rs new file mode 100644 index 00000000..c20b84db --- /dev/null +++ b/crates/cellule-runtime/src/node/bundle/closure.rs @@ -0,0 +1,144 @@ +//! Materialized checkpoints and complete issued-range closure. +use super::*; +use crate::control::authority::CellAuthority; +use crate::node::log::CellIssuedRange; +use crate::node::{NodeDirectory, VersionedNodeAdvertisement}; + +impl NodeDirectory { + /// Advances a Cell's authenticated reconstruction base only after its exact + /// materialized root CAS at a complete capture boundary. Hot later ranges + /// retain their locators while the covered prefix is released, without crediting + /// an upload, approximate watermark, or root that ends inside a command group. + pub async fn checkpoint_bundle_cell( + &self, + observed: &VersionedNodeAdvertisement, + authority: &CellAuthority, + pin: BundleBindingRef, + limits: cellule_ltx::Limits, + now_ms: i64, + ) -> Result { + let head = observed + .advertisement + .bundle + .ok_or(Error::Node("bundle lane is absent"))?; + let mut catalog = load_catalog(&self.layout, pin.session, head).await?; + if pin.epoch != catalog.epoch { + return Err(Error::Fenced); + } + let binding = catalog.binding_mut(pin.digest)?; + if authority.layout().application_id() != binding.application.as_bytes() + || authority.layout().node_path(pin.session.as_bytes()) + != self.layout.node_path(pin.session.as_bytes()) + || authority.layout().immutable_cache_identity() + != self.layout.immutable_cache_identity() + { + return Err(Error::Fenced); + } + let current = authority + .load(binding.control.cell) + .await? + .ok_or(Error::Fenced)?; + let control = current.value(); + if control.bundle_binding != Some(pin) + || control.epoch != binding.control.epoch + || control.incarnation != binding.control.incarnation + || control.code != binding.control.code + || control.schema != binding.control.schema + { + return Err(Error::PendingPublication); + } + let root = control.ltx_root().ok_or(Error::PendingPublication)?; + let frames = verify_binding(&self.layout, pin.session, head.epoch, binding, limits).await?; + let prefix = checkpoint_prefix(binding, &frames, &root)?; + if prefix == 0 && Some(root) == binding.control.ltx_root() { + return if binding.locators.is_empty() { + Ok(observed.clone()) + } else { + Err(Error::PendingPublication) + }; + } + binding.control = control.clone(); + verify_base(&self.layout, binding, limits).await?; + binding.locators.drain(..prefix); + if binding.locators.is_empty() { + binding.first_commit = binding.selected_commit; + } + let prepared = self.upload_catalog(Some(head), catalog, &[]).await?; + self.select_catalog(observed, &prepared, now_ms).await + } + + /// Freezes the original writer's complete issued endpoint from its native + /// assignment barrier. A selected-only endpoint cannot close prior Fleet ACKs. + pub async fn begin_bundle_close( + &self, + observed: &VersionedNodeAdvertisement, + pin: BundleBindingRef, + issued: CellIssuedRange, + now_ms: i64, + ) -> Result { + let head = observed + .advertisement + .bundle + .ok_or(Error::Node("bundle lane is absent"))?; + let mut catalog = load_catalog(&self.layout, pin.session, head).await?; + if pin.epoch != catalog.epoch { + return Err(Error::Fenced); + } + let binding = catalog.binding_mut(pin.digest)?; + let scope = issued.scope(); + if issued.leader_session() != pin.session + || issued.log_epoch() != pin.epoch + || scope.application != binding.application + || scope.cell != binding.control.cell + || scope.incarnation != binding.control.incarnation + || scope.cell_epoch != binding.control.epoch + || binding.phase == BindingPhase::Closed + { + return Err(Error::Fenced); + } + let terminal = ( + issued.last_node_sequence(), + issued.commit_sequence(), + issued.position(), + ); + if binding.terminal.is_some_and(|before| before != terminal) { + return Err(Error::Fenced); + } + binding.phase = BindingPhase::Closing; + binding.terminal = Some(terminal); + let prepared = self.upload_catalog(Some(head), catalog, &[]).await?; + self.select_catalog(observed, &prepared, now_ms).await + } + + /// Completes closure only at the exact frozen issued endpoint. Selection may + /// continue for hot siblings and for previously assigned rows while Closing. + pub async fn finish_bundle_close( + &self, + observed: &VersionedNodeAdvertisement, + pin: BundleBindingRef, + now_ms: i64, + ) -> Result { + let head = observed + .advertisement + .bundle + .ok_or(Error::Node("bundle lane is absent"))?; + let mut catalog = load_catalog(&self.layout, pin.session, head).await?; + let binding = catalog.binding_mut(pin.digest)?; + if pin.epoch != head.epoch { + return Err(Error::Fenced); + } + if binding.phase == BindingPhase::Open + || binding.terminal + != Some(( + binding.selected_sequence, + binding.selected_commit, + binding.selected_position, + )) + { + return Err(Error::PendingPublication); + } + binding.phase = BindingPhase::Closed; + let prepared = self.upload_catalog(Some(head), catalog, &[]).await?; + self.select_catalog(observed, &prepared, now_ms).await + } +} diff --git a/crates/cellule-runtime/src/node/bundle/codec.rs b/crates/cellule-runtime/src/node/bundle/codec.rs new file mode 100644 index 00000000..34f3a3e7 --- /dev/null +++ b/crates/cellule-runtime/src/node/bundle/codec.rs @@ -0,0 +1,220 @@ +//! Bounded binary catalog and verbatim native frame extents. +use super::*; +use crate::codec::{BoundedDecoder, BoundedEncoder, read_fixed}; + +const MAGIC: &[u8] = b"CNB1"; + +fn position_write(e: &mut BoundedEncoder, p: cellule_ltx::Position) -> Result<()> { + e.write_u64(p.txid)?; + e.write_u64(p.checksum)?; + Ok(()) +} +fn position_read(d: &mut BoundedDecoder<'_>) -> Result { + Ok(cellule_ltx::Position { + txid: d.read_u64()?, + checksum: d.read_u64()?, + }) +} + +fn metadata(catalog: &Catalog, frames: usize) -> Result { + catalog.validate()?; + let mut e = BoundedEncoder::new(MAX_BUNDLE_BYTES as u32)?; + e.write_bytes(MAGIC)?; + e.write_bytes(catalog.session.as_bytes())?; + e.write_u64(catalog.epoch)?; + e.write_bool(catalog.predecessor.is_some())?; + if let Some(digest) = catalog.predecessor { + e.write_bytes(digest.as_bytes())?; + } + e.write_u64(catalog.selected_through)?; + e.write_count(catalog.bindings.len())?; + for binding in &catalog.bindings { + e.write_bytes(binding.application.as_bytes())?; + e.write_u64(binding.first_commit)?; + e.write_bytes(&binding.control.encode()?)?; + e.write_u8(match binding.phase { + BindingPhase::Open => 0, + BindingPhase::Closing => 1, + BindingPhase::Closed => 2, + })?; + e.write_bool(binding.terminal.is_some())?; + if let Some((sequence, commit, position)) = binding.terminal { + e.write_u64(sequence)?; + e.write_u64(commit)?; + position_write(&mut e, position)?; + } + e.write_u64(binding.selected_sequence)?; + e.write_u64(binding.selected_commit)?; + position_write(&mut e, binding.selected_position)?; + e.write_count(binding.locators.len())?; + for locator in &binding.locators { + e.write_bool(locator.object.is_some())?; + if let Some(digest) = locator.object { + e.write_bytes(digest.as_bytes())?; + } + e.write_u64(locator.offset)?; + e.write_u64(locator.bytes)?; + e.write_bytes(locator.frame_digest.as_bytes())?; + } + } + e.write_count(frames)?; + Ok(e) +} + +pub(super) fn encode( + catalog: &mut Catalog, + frames: &[cellule_ltx::VerifiedNodeFrame], +) -> Result { + if frames.len() > MAX_FRAMES { + return Err(Error::Capacity("bundle frame count")); + } + let mut offset = metadata(catalog, frames.len())?.finish().len() as u64; + for frame in frames { + offset = offset + .checked_add(4) + .ok_or(Error::Capacity("bundle extent"))?; + let digest = Digest::from_bytes(frame.digest()); + let matches = catalog + .bindings + .iter_mut() + .flat_map(|binding| &mut binding.locators) + .filter(|locator| locator.object.is_none() && locator.frame_digest == digest) + .map(|locator| { + locator.offset = offset; + locator.bytes = frame.encoded().len() as u64; + }) + .count(); + if matches != 1 { + return Err(Error::Node("bundle frame locator is not unique")); + } + offset = offset + .checked_add(frame.encoded().len() as u64) + .ok_or(Error::Capacity("bundle bytes"))?; + } + let mut e = metadata(catalog, frames.len())?; + for frame in frames { + e.write_bytes(frame.encoded())?; + } + Ok(Bytes::from(e.finish())) +} + +pub(super) fn decode(body: &Bytes) -> Result { + let mut d = BoundedDecoder::new(body, MAX_BUNDLE_BYTES as u32)?; + if d.read_bytes()? != MAGIC { + return Err(Error::Node("unsupported node bundle format")); + } + let session = SessionId::from_bytes(read_fixed(&mut d, "node bundle fixed field length")?); + let epoch = d.read_u64()?; + let predecessor = if d.read_bool()? { + Some(Digest::from_bytes(read_fixed( + &mut d, + "node bundle fixed field length", + )?)) + } else { + None + }; + let selected_through = d.read_u64()?; + let count = d.read_count()?; + if count > MAX_BINDINGS { + return Err(Error::Capacity("bundle binding count")); + } + let mut bindings = Vec::with_capacity(count); + for _ in 0..count { + let application = crate::identity::ApplicationId::from_bytes(read_fixed( + &mut d, + "bundle application length", + )?); + let first_commit = d.read_u64()?; + let control = Control::decode(d.read_bytes()?)?; + let phase = match d.read_u8()? { + 0 => BindingPhase::Open, + 1 => BindingPhase::Closing, + 2 => BindingPhase::Closed, + _ => return Err(Error::Node("invalid bundle binding phase")), + }; + let terminal = if d.read_bool()? { + Some((d.read_u64()?, d.read_u64()?, position_read(&mut d)?)) + } else { + None + }; + let selected_sequence = d.read_u64()?; + let selected_commit = d.read_u64()?; + let selected_position = position_read(&mut d)?; + let count = d.read_count()?; + if count > MAX_LOCATORS { + return Err(Error::Capacity("bundle locator count")); + } + let mut locators = Vec::with_capacity(count); + for _ in 0..count { + locators.push(Locator { + object: if d.read_bool()? { + Some(Digest::from_bytes(read_fixed( + &mut d, + "node bundle fixed field length", + )?)) + } else { + None + }, + offset: d.read_u64()?, + bytes: d.read_u64()?, + frame_digest: Digest::from_bytes(read_fixed( + &mut d, + "node bundle fixed field length", + )?), + }); + } + bindings.push(Binding { + application, + first_commit, + control, + phase, + terminal, + selected_sequence, + selected_commit, + selected_position, + locators, + }); + } + let count = d.read_count()?; + if count > MAX_FRAMES { + return Err(Error::Capacity("bundle frame count")); + } + let mut local = Vec::with_capacity(count); + for _ in 0..count { + let frame = d.read_bytes()?; + // This offset comes from the borrowed input, rather than trusting an + // encoded locator or scanning other node objects for a matching row. + let offset = frame.as_ptr() as usize - body.as_ptr() as usize; + local.push(( + offset as u64, + frame.len() as u64, + Digest::from_bytes(*blake3::hash(frame).as_bytes()), + )); + } + d.finish()?; + let catalog = Catalog { + session, + epoch, + predecessor, + selected_through, + bindings, + }; + catalog.validate()?; + let locators: Vec<_> = catalog + .bindings + .iter() + .flat_map(|binding| &binding.locators) + .filter(|locator| locator.object.is_none()) + .map(|locator| (locator.offset, locator.bytes, locator.frame_digest)) + .collect(); + if locators.len() != local.len() + || local + .iter() + .any(|extent| locators.iter().filter(|locator| *locator == extent).count() != 1) + { + return Err(Error::Node( + "bundle local extents differ from complete manifest", + )); + } + Ok(catalog) +} diff --git a/crates/cellule-runtime/src/node/bundle/mod.rs b/crates/cellule-runtime/src/node/bundle/mod.rs new file mode 100644 index 00000000..92a20e6f --- /dev/null +++ b/crates/cellule-runtime/src/node/bundle/mod.rs @@ -0,0 +1,250 @@ +//! Verified node-wide coverage and bounded per-Cell reconstruction locators. +//! +//! Upload is a proposal. The canonical node-record CAS selects its complete +//! binding catalog and native range together. Cell authority pins each binding +//! and refuses departure until its complete issued range has a materialized root. +//! +//! The caller owns host admission and the original node lease. This selection +//! helper operates on complete captures assigned by the canonical shipper; +//! returned reconstruction proofs do not enable the actor's command response. +//! +//! ```no_run +//! use cellule_runtime::node::{NodeDirectory, VersionedNodeAdvertisement}; +//! use cellule_runtime::node::bundle::BundleCoverageProof; +//! use cellule_runtime::node::lease::NodeLeaseGuard; +//! use cellule_runtime::node::log::AssignedCommitRange; +//! +//! async fn select_complete_captures( +//! directory: &NodeDirectory, +//! observed: &VersionedNodeAdvertisement, +//! lease: &NodeLeaseGuard, +//! frames: &[cellule_ltx::VerifiedNodeFrame], +//! assignments: &[AssignedCommitRange], +//! now_ms: i64, +//! ) -> cellule_runtime::Result<(VersionedNodeAdvertisement, Vec)> { +//! let proposal = directory +//! .prepare_node_bundle(observed, frames, assignments, now_ms).await?; +//! directory.select_node_bundle( +//! observed, &proposal, lease, cellule_ltx::Limits::default(), now_ms, +//! ).await +//! } +//! ``` + +use crate::control::{BundleBindingRef, Control}; +use crate::identity::{Digest, SessionId}; +use crate::{Error, Result}; +use bytes::Bytes; + +mod binding; +mod closure; +mod codec; +mod proof; +mod selection; +use proof::{checkpoint_prefix, verify_base, verify_binding}; +use store::load_catalog; +pub(crate) mod store; +#[cfg(test)] +mod tests; + +pub(crate) const MAX_BUNDLE_BYTES: u64 = 4 << 20; +const MAX_BINDINGS: usize = 4_096; +const MAX_LOCATORS: usize = 32; +const MAX_FRAMES: usize = 64; +// Verification retains at most one Cell suffix, independently of the number +// of historical objects referenced by its locators. +const MAX_SUFFIX_BYTES: u64 = MAX_BUNDLE_BYTES; +const MAX_BASE_OBJECTS: usize = 65_536; + +/// Canonical node-record pointer; observing it alone grants no response proof. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub struct NodeBundleHead { + pub(crate) epoch: u64, + pub(crate) digest: Digest, + pub(crate) selected_through: u64, +} +impl NodeBundleHead { + /// Publication lane epoch. + pub const fn epoch(&self) -> u64 { + self.epoch + } + /// Digest of the complete immutable catalog and range. + pub const fn digest(&self) -> Digest { + self.digest + } + /// Exact contiguous selected native range endpoint. + pub const fn selected_through(&self) -> u64 { + self.selected_through + } + pub(crate) fn validate(&self) -> Result<()> { + if self.epoch == 0 || self.digest.as_bytes().iter().all(|byte| *byte == 0) { + return Err(Error::Node("invalid node bundle head")); + } + Ok(()) + } +} + +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +enum BindingPhase { + Open, + Closing, + Closed, +} + +#[derive(Clone, Debug, PartialEq, Eq)] +struct Locator { + // None in a newly encoded manifest refers to that very immutable object. + // Resolving it on load avoids a self-referential digest in the byte format. + object: Option, + offset: u64, + bytes: u64, + frame_digest: Digest, +} + +#[derive(Clone, Debug, PartialEq, Eq)] +struct Binding { + application: crate::identity::ApplicationId, + first_commit: u64, + control: Control, + phase: BindingPhase, + terminal: Option<(u64, u64, cellule_ltx::Position)>, + selected_sequence: u64, + selected_commit: u64, + selected_position: cellule_ltx::Position, + locators: Vec, +} + +#[derive(Clone, Debug, PartialEq, Eq)] +struct Catalog { + session: SessionId, + epoch: u64, + predecessor: Option, + selected_through: u64, + bindings: Vec, +} + +/// Uploaded exact proposal. It cannot release an ACK or prune a capture. +pub struct PreparedNodeBundle { + original: Option, + catalog: Catalog, + body: Bytes, + head: NodeBundleHead, +} + +/// Selected, dependency-verified coverage of one exact Cell writer. +/// +/// The bounded locators retain no frame bodies. Creation is restricted to a +/// successful/reconciled canonical node CAS with prior complete range verification. +pub struct BundleCoverageProof { + pin: BundleBindingRef, + binding: Binding, + head: NodeBundleHead, + session: SessionId, +} +impl BundleCoverageProof { + /// Original Cell authority pin. + pub fn binding(&self) -> BundleBindingRef { + self.pin + } + /// Exact covered logical command endpoint. + pub const fn commit_sequence(&self) -> u64 { + self.binding.selected_commit + } + /// Exact covered SQLite position. + pub const fn position(&self) -> cellule_ltx::Position { + self.binding.selected_position + } + /// Number of bounded authenticated frame locators retained by the proof. + pub fn locator_count(&self) -> usize { + self.binding.locators.len() + } + /// Exact immutable base required for reconstruction. + pub fn base(&self) -> Result { + self.binding + .control + .ltx_root() + .ok_or(Error::Node("bundle binding has no base")) + } +} + +impl Catalog { + fn validate(&self) -> Result<()> { + if self.epoch == 0 + || self.session.as_bytes().iter().all(|byte| *byte == 0) + || self.bindings.len() > MAX_BINDINGS + { + return Err(Error::Node("invalid bundle catalog bounds")); + } + let mut previous = None; + let mut scopes = std::collections::HashSet::new(); + for binding in &self.bindings { + binding.control.encode()?; + let pin = binding + .control + .bundle_binding + .ok_or(Error::Node("bundle catalog lacks Cell pin"))?; + let base = binding + .control + .ltx_root() + .ok_or(Error::Node("bundle catalog lacks base"))?; + if pin.session != self.session + || pin.epoch != self.epoch + || binding.first_commit == 0 + || binding.first_commit > binding.selected_commit + || !scopes.insert(( + binding.application, + binding.control.cell, + binding.control.incarnation, + binding.control.epoch, + )) + || previous.is_some_and(|digest| digest >= *pin.digest.as_bytes()) + || binding.locators.len() > MAX_LOCATORS + || binding.selected_sequence > self.selected_through + || binding.selected_commit < base.commit_sequence + || binding.selected_position.txid < base.position.txid + || ((binding.selected_commit == base.commit_sequence) + != binding.locators.is_empty()) + || (binding.locators.is_empty() && binding.selected_position != base.position) + || binding + .locators + .iter() + .try_fold(0_u64, |bytes, locator| bytes.checked_add(locator.bytes)) + .is_none_or(|bytes| bytes > MAX_SUFFIX_BYTES) + || binding.locators.iter().any(|locator| { + locator.bytes == 0 + || locator.bytes > MAX_BUNDLE_BYTES + || locator + .offset + .checked_add(locator.bytes) + .is_none_or(|end| end > MAX_BUNDLE_BYTES) + }) + { + return Err(Error::Node("invalid bundle binding coverage")); + } + match (binding.phase, binding.terminal) { + (BindingPhase::Open, None) => {} + (BindingPhase::Closing, Some((sequence, commit, position))) + if sequence >= binding.selected_sequence + && commit >= binding.selected_commit + && position.txid >= binding.selected_position.txid => {} + (BindingPhase::Closed, Some((sequence, commit, position))) + if sequence == binding.selected_sequence + && commit == binding.selected_commit + && position == binding.selected_position => {} + _ => return Err(Error::Node("invalid bundle binding closure")), + } + previous = Some(*pin.digest.as_bytes()); + } + Ok(()) + } + fn binding_mut(&mut self, digest: Digest) -> Result<&mut Binding> { + self.bindings + .iter_mut() + .find(|binding| { + binding + .control + .bundle_binding + .is_some_and(|pin| pin.digest == digest) + }) + .ok_or(Error::Node("bundle binding missing")) + } +} diff --git a/crates/cellule-runtime/src/node/bundle/proof.rs b/crates/cellule-runtime/src/node/bundle/proof.rs new file mode 100644 index 00000000..515232bc --- /dev/null +++ b/crates/cellule-runtime/src/node/bundle/proof.rs @@ -0,0 +1,243 @@ +//! Selected suffix verification and exact reconstruction overlays. +use super::*; +use crate::control::authority::{CellAuthority, VersionedControl}; +use crate::node::NodeDirectory; +use crate::node::directory::NodeRecord; + +impl NodeDirectory { + /// Reopens one authority-pinned selected Cell suffix from canonical origin. + /// This is a reconstruction capability, including after the original boot + /// was fenced. It grants no writer lease, command ACK, or complete follower + /// suffix: issued Fleet ranges above the selected head still require drain. + pub async fn load_bundle_coverage( + &self, + authority: &CellAuthority, + control: &VersionedControl, + limits: cellule_ltx::Limits, + ) -> Result { + let value = control.value(); + let pin = value.bundle_binding.ok_or(Error::PendingPublication)?; + if authority.layout().node_path(pin.session.as_bytes()) + != self.layout.node_path(pin.session.as_bytes()) + || authority.layout().immutable_cache_identity() + != self.layout.immutable_cache_identity() + { + return Err(Error::Fenced); + } + let (body, _) = self + .layout + .store() + .get_with_etag_bounded( + &self.layout.node_path(pin.session.as_bytes()), + crate::node::MAX_NODE_BYTES, + ) + .await?; + let record = NodeRecord::decode_canonical(&body)?; + let (session, head) = match record { + NodeRecord::Advertisement(node) => (node.session, node.bundle), + NodeRecord::Tombstone(node) => (node.session, node.bundle), + }; + if session != pin.session { + return Err(Error::Fenced); + } + let head = head.ok_or(Error::PendingPublication)?; + let mut catalog = load_catalog(&self.layout, session, head).await?; + let binding = catalog.binding_mut(pin.digest)?.clone(); + if head.epoch != pin.epoch + || binding.control.bundle_binding != Some(pin) + || binding.application.as_bytes() != authority.layout().application_id() + || binding.control.cell != value.cell + || binding.control.incarnation != value.incarnation + || binding.control.epoch != value.epoch + || binding.control.code != value.code + || binding.control.schema != value.schema + { + return Err(Error::Fenced); + } + let frames = verify_binding(&self.layout, session, head.epoch, &binding, limits).await?; + let current = value.ltx_root().ok_or(Error::Fenced)?; + checkpoint_prefix(&binding, &frames, ¤t)?; + Ok(BundleCoverageProof { + pin, + binding, + head, + session, + }) + } +} + +pub(super) async fn verify_binding( + layout: &cellule_ltx::CellStorageLayout, + session: SessionId, + epoch: u64, + binding: &Binding, + limits: cellule_ltx::Limits, +) -> Result> { + verify_base(layout, binding, limits).await?; + let mut position = binding + .control + .ltx_root() + .ok_or(Error::Node("bundle base absent"))? + .position; + let mut commit = binding + .control + .root + .as_ref() + .ok_or(Error::Node("bundle base absent"))? + .commit_sequence; + let mut frames = Vec::with_capacity(binding.locators.len()); + let mut sequence = 0; + let mut first_commit = commit; + for locator in &binding.locators { + let object = locator + .object + .ok_or(Error::Node("bundle locator is unresolved"))?; + let end = locator + .offset + .checked_add(locator.bytes) + .ok_or(Error::Node("bundle locator overflow"))?; + let bytes = layout + .store() + .range_get( + &layout.node_coverage_bundle_path(session.as_bytes(), epoch, object.as_bytes()), + locator.offset..end, + ) + .await?; + if bytes.len() as u64 != locator.bytes + || *blake3::hash(&bytes).as_bytes() != *locator.frame_digest.as_bytes() + { + return Err(Error::Node("bundle frame digest differs")); + } + let frame = cellule_ltx::inspect_node_frame(bytes, limits)?; + let scope = frame.scope(); + if scope.leader_session != *session.as_bytes() + || scope.log_epoch != epoch + || scope.application != *binding.application.as_bytes() + || scope.cell != *binding.control.cell.as_bytes() + || scope.incarnation != *binding.control.incarnation.as_bytes() + || scope.cell_epoch != binding.control.epoch + || scope.node_sequence <= sequence + || (scope.commit_sequence == commit && frame.first_commit_sequence() != first_commit) + || (scope.commit_sequence != commit + && commit.checked_add(1) != Some(frame.first_commit_sequence())) + || position.txid.checked_add(1) != Some(frame.segment().min_txid) + || position.checksum != frame.segment().pre_checksum + { + return Err(Error::Node("bundle locator violates exact Cell range")); + } + first_commit = frame.first_commit_sequence(); + position = frame.segment().position(); + commit = scope.commit_sequence; + sequence = scope.node_sequence; + frames.push(frame); + } + if position != binding.selected_position + || commit != binding.selected_commit + || (!binding.locators.is_empty() && sequence != binding.selected_sequence) + { + return Err(Error::Node("bundle proof endpoint differs")); + } + Ok(frames) +} + +pub(super) async fn verify_base( + layout: &cellule_ltx::CellStorageLayout, + binding: &Binding, + limits: cellule_ltx::Limits, +) -> Result<()> { + let base = binding + .control + .ltx_root() + .ok_or(Error::Node("bundle base absent"))?; + let replica = cellule_ltx::CellReplica::new( + layout.for_application(*binding.application.as_bytes()), + *binding.control.cell.as_bytes(), + *binding.control.incarnation.as_bytes(), + limits, + )?; + // Reconstructability requires origin dependencies, even if metadata was + // authenticated earlier in this process. A cached root is not availability. + replica + .reachable_objects_bounded(&base, MAX_BASE_OBJECTS) + .await?; + Ok(()) +} + +impl BundleCoverageProof { + /// Revalidates every bounded locator and builds an exact overlay for normal + /// root preparation. This runs independently of the selecting bundle CAS. + pub async fn recovery_overlay( + &self, + layout: &cellule_ltx::CellStorageLayout, + limits: cellule_ltx::Limits, + ) -> Result { + self.recovery_overlay_from(layout, limits, self.base()?) + .await + } + + pub(crate) async fn recovery_overlay_from( + &self, + layout: &cellule_ltx::CellStorageLayout, + limits: cellule_ltx::Limits, + base: cellule_ltx::RootRef, + ) -> Result { + if layout.application_id() != self.binding.application.as_bytes() { + return Err(Error::Fenced); + } + let frames = + verify_binding(layout, self.session, self.head.epoch, &self.binding, limits).await?; + let skip = checkpoint_prefix(&self.binding, &frames, &base)?; + let entries = frames[skip..] + .iter() + .map(|frame| { + cellule_ltx::bundle::BundleEntry::for_cell( + *self.binding.control.cell.as_bytes(), + *self.binding.control.incarnation.as_bytes(), + frame.segment().clone(), + frame.body().to_vec(), + ) + }) + .collect(); + let bundle = cellule_ltx::bundle::Bundle::encode(entries, limits)?; + Ok(cellule_ltx::RecoveryOverlay::new( + base, + bundle, + self.position(), + self.commit_sequence(), + )) + } +} + +/// An exact complete capture boundary may checkpoint a prefix while later +/// captures continue selection. A row's final command number alone cannot +/// checkpoint inside a multi-frame physical capture. +pub(super) fn checkpoint_prefix( + binding: &Binding, + frames: &[cellule_ltx::VerifiedNodeFrame], + root: &cellule_ltx::RootRef, +) -> Result { + let base = binding + .control + .ltx_root() + .ok_or(Error::Node("bundle base absent"))?; + if root.cell != base.cell || root.incarnation != base.incarnation { + return Err(Error::Fenced); + } + if root.commit_sequence == base.commit_sequence && root.position == base.position { + return Ok(0); + } + for (index, frame) in frames.iter().enumerate() { + if frame.scope().commit_sequence == root.commit_sequence + && frame.segment().position() == root.position + { + if frames + .get(index + 1) + .is_some_and(|next| next.scope().commit_sequence == root.commit_sequence) + { + return Err(Error::Node("bundle checkpoint splits an assigned capture")); + } + return Ok(index + 1); + } + } + Err(Error::PendingPublication) +} diff --git a/crates/cellule-runtime/src/node/bundle/selection.rs b/crates/cellule-runtime/src/node/bundle/selection.rs new file mode 100644 index 00000000..8c655faf --- /dev/null +++ b/crates/cellule-runtime/src/node/bundle/selection.rs @@ -0,0 +1,150 @@ +//! Exact native assignment verification and shared authority selection. +use super::*; +use crate::node::lease::NodeLeaseGuard; +use crate::node::{NodeDirectory, VersionedNodeAdvertisement}; + +impl NodeDirectory { + /// Verifies a contiguous complete native range and uploads one proposal for + /// every participating Cell. Neither upload nor this value grants an ACK. + pub async fn prepare_node_bundle( + &self, + observed: &VersionedNodeAdvertisement, + frames: &[cellule_ltx::VerifiedNodeFrame], + assignments: &[crate::node::log::AssignedCommitRange], + now_ms: i64, + ) -> Result { + self.validate(&observed.advertisement, now_ms)?; + let head = observed + .advertisement + .bundle + .ok_or(Error::Node("bundle lane is absent"))?; + let mut catalog = load_catalog(&self.layout, observed.advertisement.session, head).await?; + if frames.is_empty() || frames.len() > MAX_FRAMES { + return Err(Error::Capacity("bundle frame count")); + } + let mut consumed = 0_usize; + for assignment in assignments { + let count = usize::try_from( + assignment.ticket().last_sequence() - assignment.ticket().first_sequence() + 1, + ) + .map_err(|_| Error::Capacity("assigned range count"))?; + let end = consumed + .checked_add(count) + .ok_or(Error::Capacity("assigned range count"))?; + assignment.verify( + frames + .get(consumed..end) + .ok_or(Error::Node("bundle omits assigned capture"))?, + )?; + consumed = end; + } + if consumed != frames.len() { + return Err(Error::Node("bundle contains unassigned native frames")); + } + for frame in frames { + let scope = frame.scope(); + if scope.leader_session != *catalog.session.as_bytes() + || scope.log_epoch != catalog.epoch + || catalog.selected_through.checked_add(1) != Some(scope.node_sequence) + { + return Err(Error::Node( + "bundle native range is not contiguous in its lane", + )); + } + let binding = catalog + .bindings + .iter_mut() + .find(|binding| { + binding.application.as_bytes() == &scope.application + && binding.control.cell.as_bytes() == &scope.cell + && binding.control.incarnation.as_bytes() == &scope.incarnation + && binding.control.epoch == scope.cell_epoch + }) + .ok_or(Error::Node("bundle row has no enrolled Cell binding"))?; + if binding.phase == BindingPhase::Closed + || binding.locators.len() >= MAX_LOCATORS + || binding + .terminal + .is_some_and(|(sequence, commit, position)| { + scope.node_sequence > sequence + || scope.commit_sequence > commit + || frame.segment().max_txid > position.txid + }) + { + return Err(Error::PendingPublication); + } + let continuation = binding.selected_commit == scope.commit_sequence; + if (continuation && binding.first_commit != frame.first_commit_sequence()) + || (!continuation + && binding.selected_commit.checked_add(1) + != Some(frame.first_commit_sequence())) + || binding.selected_position.txid.checked_add(1) != Some(frame.segment().min_txid) + || binding.selected_position.checksum != frame.segment().pre_checksum + { + return Err(Error::Node( + "bundle Cell transaction or command range has a gap", + )); + } + binding.first_commit = frame.first_commit_sequence(); + binding.selected_commit = scope.commit_sequence; + binding.selected_position = frame.segment().position(); + binding.selected_sequence = scope.node_sequence; + binding.locators.push(Locator { + object: None, + offset: 0, + bytes: frame.encoded().len() as u64, + frame_digest: Digest::from_bytes(frame.digest()), + }); + catalog.selected_through = scope.node_sequence; + } + self.upload_catalog(Some(head), catalog, frames).await + } + + /// Selects exactly the uploaded catalog/range after I/O, under the original + /// live node lease. A changed catalog/head forces a new preparation; an + /// unchanged head may rebase against heartbeat-only CAS conflicts. + pub async fn select_node_bundle( + &self, + observed: &VersionedNodeAdvertisement, + prepared: &PreparedNodeBundle, + lease: &NodeLeaseGuard, + limits: cellule_ltx::Limits, + now_ms: i64, + ) -> Result<(VersionedNodeAdvertisement, Vec)> { + lease.check()?; + let catalog = load_catalog(&self.layout, prepared.catalog.session, prepared.head).await?; + let mut proofs = Vec::new(); + for binding in catalog.bindings { + if !binding + .locators + .iter() + .any(|locator| locator.object == Some(prepared.head.digest)) + { + continue; + } + verify_binding( + &self.layout, + catalog.session, + catalog.epoch, + &binding, + limits, + ) + .await?; + let pin = binding + .control + .bundle_binding + .ok_or(Error::Node("selected binding lacks pin"))?; + proofs.push(BundleCoverageProof { + pin, + binding, + head: prepared.head, + session: catalog.session, + }); + } + lease.check()?; + let selected = self.select_catalog(observed, prepared, now_ms).await?; + // Expiry during verification must not revive the original writer. + lease.check()?; + Ok((selected, proofs)) + } +} diff --git a/crates/cellule-runtime/src/node/bundle/store.rs b/crates/cellule-runtime/src/node/bundle/store.rs new file mode 100644 index 00000000..94082534 --- /dev/null +++ b/crates/cellule-runtime/src/node/bundle/store.rs @@ -0,0 +1,200 @@ +//! Canonical immutable catalog I/O and departure guards. +use super::*; +use crate::node::directory::NodeRecord; +use crate::node::{NodeDirectory, VersionedNodeAdvertisement}; + +impl NodeDirectory { + pub(super) async fn upload_catalog( + &self, + original: Option, + mut catalog: Catalog, + frames: &[cellule_ltx::VerifiedNodeFrame], + ) -> Result { + if self.layout.store().staging_write_prefix().is_some() { + return Err(Error::Node( + "bundle selection requires canonical origin storage", + )); + } + catalog.predecessor = original.map(|head| head.digest); + let body = codec::encode(&mut catalog, frames)?; + // Decode the very bytes sent to the provider before selecting them. + if codec::decode(&body)? != catalog { + return Err(Error::Node("bundle manifest self-verification differs")); + } + let head = NodeBundleHead { + epoch: catalog.epoch, + digest: Digest::from_bytes(*blake3::hash(&body).as_bytes()), + selected_through: catalog.selected_through, + }; + self.layout + .store() + .put_exact( + &self.layout.node_coverage_bundle_path( + catalog.session.as_bytes(), + catalog.epoch, + head.digest.as_bytes(), + ), + body.clone(), + ) + .await?; + Ok(PreparedNodeBundle { + original, + catalog, + body, + head, + }) + } + + pub(super) async fn select_catalog( + &self, + observed: &VersionedNodeAdvertisement, + prepared: &PreparedNodeBundle, + now_ms: i64, + ) -> Result { + let mut base = observed.clone(); + if *blake3::hash(&prepared.body).as_bytes() != *prepared.head.digest.as_bytes() { + return Err(Error::Node("bundle proposal digest differs")); + } + for _ in 0..4 { + self.validate(&base.advertisement, now_ms)?; + if base.advertisement.session != prepared.catalog.session + || base.advertisement.log.as_ref().is_some_and(|log| { + log.epoch() != prepared.head.epoch + || log.phase() != crate::node::log_state::NodeLogPhase::Open + }) + { + return Err(Error::Fenced); + } + if base.advertisement.bundle == Some(prepared.head) { + return Ok(base); + } + if base.advertisement.bundle != prepared.original { + return Err(Error::Fenced); + } + let mut next = base.advertisement.clone(); + next.bundle = Some(prepared.head); + next.generation = next + .generation + .checked_add(1) + .ok_or(Error::Node("node generation overflow"))?; + match self.update_advertisement(&base, next, now_ms).await { + Ok(selected) => return Ok(selected), + Err(source) => match self.load(prepared.catalog.session, now_ms).await { + Ok(Some(current)) + if current.advertisement.bundle == Some(prepared.head) + || current.advertisement.bundle == prepared.original => + { + base = current + } + _ => return Err(source), + }, + } + } + Err(Error::Node("bundle selection CAS contention")) + } +} + +pub(super) async fn load_catalog( + layout: &cellule_ltx::CellStorageLayout, + session: SessionId, + head: NodeBundleHead, +) -> Result { + let (body, _) = layout + .store() + .get_with_etag_bounded( + &layout.node_coverage_bundle_path( + session.as_bytes(), + head.epoch, + head.digest.as_bytes(), + ), + MAX_BUNDLE_BYTES, + ) + .await?; + if *blake3::hash(&body).as_bytes() != *head.digest.as_bytes() { + return Err(Error::Node("bundle catalog digest differs")); + } + let mut catalog = codec::decode(&body)?; + if catalog.session != session + || catalog.epoch != head.epoch + || catalog.selected_through != head.selected_through + { + return Err(Error::Node("bundle catalog scope differs")); + } + for locator in catalog + .bindings + .iter_mut() + .flat_map(|binding| &mut binding.locators) + { + if locator.object.is_none() { + locator.object = Some(head.digest); + } + } + Ok(catalog) +} +/// Shared by every lower-level departure CAS. A caller cannot bypass closure +/// by using CellAuthority directly instead of the live actor's drain API. +pub(crate) async fn ensure_departure( + layout: &cellule_ltx::CellStorageLayout, + control: &Control, +) -> Result<()> { + let Some(pin) = control.bundle_binding else { + return Ok(()); + }; + let (body, _) = layout + .store() + .get_with_etag_bounded( + &layout.node_path(pin.session.as_bytes()), + crate::node::MAX_NODE_BYTES, + ) + .await?; + let record = NodeRecord::decode_canonical(&body)?; + let (session, head) = match record { + NodeRecord::Advertisement(node) => (node.session, node.bundle), + NodeRecord::Tombstone(node) => (node.session, node.bundle), + }; + if session != pin.session { + return Err(Error::Fenced); + } + let head = head.ok_or(Error::PendingPublication)?; + let mut catalog = load_catalog(layout, pin.session, head).await?; + if pin.epoch != catalog.epoch { + return Err(Error::Fenced); + } + let binding = catalog.binding_mut(pin.digest)?; + let root = control.ltx_root().ok_or(Error::PendingPublication)?; + if binding.phase != BindingPhase::Closed + || !binding.locators.is_empty() + || binding.application.as_bytes() != layout.application_id() + || binding.control.cell != control.cell + || binding.control.incarnation != control.incarnation + || binding.control.epoch != control.epoch + || root.commit_sequence != binding.selected_commit + || root.position != binding.selected_position + || binding.control.ltx_root() != Some(root) + { + return Err(Error::PendingPublication); + } + Ok(()) +} + +/// An immutable terminal catalog can outlive its boot record. Maintenance may +/// only treat that boot as drained after every binding checkpoint removed its +/// reconstruction dependencies under the exact materialized Cell root CAS. +pub(crate) async fn ensure_session_drained( + layout: &cellule_ltx::CellStorageLayout, + session: SessionId, + head: Option, +) -> Result<()> { + let Some(head) = head else { + return Ok(()); + }; + let catalog = load_catalog(layout, session, head).await?; + if catalog + .bindings + .iter() + .any(|binding| binding.phase != BindingPhase::Closed || !binding.locators.is_empty()) + { + return Err(Error::PendingPublication); + } + Ok(()) +} diff --git a/crates/cellule-runtime/src/node/bundle/tests/faults.rs b/crates/cellule-runtime/src/node/bundle/tests/faults.rs new file mode 100644 index 00000000..6a3df959 --- /dev/null +++ b/crates/cellule-runtime/src/node/bundle/tests/faults.rs @@ -0,0 +1,136 @@ +use super::*; +use futures_util::stream::BoxStream; +use object_store::{ + CopyOptions, GetOptions, GetResult, ListResult, MultipartUpload, ObjectMeta, ObjectStore, + PutMultipartOptions, PutOptions, PutPayload, PutResult, +}; +use std::sync::atomic::{AtomicU8, Ordering}; + +#[derive(Debug, Default)] +struct ReplyFault { + inner: InMemory, + mode: AtomicU8, +} +impl std::fmt::Display for ReplyFault { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + f.write_str("bundle-reply-fault") + } +} +fn denied() -> object_store::Error { + object_store::Error::NotSupported { + source: Box::new(std::io::Error::other("injected bundle reply failure")), + } +} +#[async_trait::async_trait] +impl ObjectStore for ReplyFault { + async fn put_opts( + &self, + path: &Path, + body: PutPayload, + opts: PutOptions, + ) -> object_store::Result { + let mode = self.mode.load(Ordering::SeqCst); + let eligible = (mode == 1 && path.as_ref().ends_with(".cnb")) + || (mode == 2 + && path.as_ref().contains("/nodes/") + && matches!(opts.mode, object_store::PutMode::Update(_))); + let lose = eligible + && self + .mode + .compare_exchange(mode, 0, Ordering::SeqCst, Ordering::SeqCst) + .is_ok(); + let result = self.inner.put_opts(path, body, opts).await?; + if lose { + return Err(denied()); + } + Ok(result) + } + async fn put_multipart_opts( + &self, + path: &Path, + opts: PutMultipartOptions, + ) -> object_store::Result> { + self.inner.put_multipart_opts(path, opts).await + } + async fn get_opts(&self, path: &Path, opts: GetOptions) -> object_store::Result { + self.inner.get_opts(path, opts).await + } + fn delete_stream( + &self, + paths: BoxStream<'static, object_store::Result>, + ) -> BoxStream<'static, object_store::Result> { + self.inner.delete_stream(paths) + } + fn list(&self, prefix: Option<&Path>) -> BoxStream<'static, object_store::Result> { + self.inner.list(prefix) + } + async fn list_with_delimiter(&self, prefix: Option<&Path>) -> object_store::Result { + self.inner.list_with_delimiter(prefix).await + } + async fn copy_opts( + &self, + from: &Path, + to: &Path, + opts: CopyOptions, + ) -> object_store::Result<()> { + self.inner.copy_opts(from, to, opts).await + } +} + +#[tokio::test] +async fn lost_immutable_reply_grants_no_proof_and_retry_reuses_exact_bytes() { + let faults = Arc::new(ReplyFault::default()); + let mut f = Fixture::with_store(faults.clone()).await; + let mut cell = f.cell(4).await; + let (_, frames, assigned) = f.append(&mut cell, 2); + faults.mode.store(1, Ordering::SeqCst); + // Even a successful origin PUT with a lost reply cannot select a range. + assert!( + f.directory + .prepare_node_bundle(&f.node, &frames, &[assigned], NOW) + .await + .is_err() + ); + assert_eq!( + f.directory + .load(SessionId::from_bytes([1; 16]), NOW) + .await + .unwrap() + .unwrap() + .advertisement() + .bundle_head(), + f.node.advertisement().bundle_head() + ); + let retry = f + .directory + .prepare_node_bundle(&f.node, &frames, &[assigned], NOW) + .await + .unwrap(); + let (_, proofs) = f + .directory + .select_node_bundle(&f.node, &retry, &f.lease, Limits::default(), NOW) + .await + .unwrap(); + assert_eq!(proofs[0].commit_sequence(), 2); +} + +#[tokio::test] +async fn lost_node_cas_reply_reconciles_only_the_exact_selected_head() { + let faults = Arc::new(ReplyFault::default()); + let mut f = Fixture::with_store(faults.clone()).await; + let mut cell = f.cell(4).await; + let (_, frames, assigned) = f.append(&mut cell, 2); + let prepared = f + .directory + .prepare_node_bundle(&f.node, &frames, &[assigned], NOW) + .await + .unwrap(); + faults.mode.store(2, Ordering::SeqCst); + let (selected, proofs) = f + .directory + .select_node_bundle(&f.node, &prepared, &f.lease, Limits::default(), NOW) + .await + .unwrap(); + assert_eq!(selected.advertisement().bundle_head(), Some(prepared.head)); + assert_eq!(proofs[0].commit_sequence(), 2); +} diff --git a/crates/cellule-runtime/src/node/bundle/tests/lifecycle.rs b/crates/cellule-runtime/src/node/bundle/tests/lifecycle.rs new file mode 100644 index 00000000..968de53b --- /dev/null +++ b/crates/cellule-runtime/src/node/bundle/tests/lifecycle.rs @@ -0,0 +1,358 @@ +use super::*; + +#[tokio::test] +async fn asynchronous_prefix_checkpoint_keeps_a_hot_selected_suffix() { + verify_hot_materialization(true).await; +} + +#[tokio::test] +async fn materializer_can_continue_from_a_newer_root_before_catalog_checkpoint() { + verify_hot_materialization(false).await; +} + +async fn verify_hot_materialization(checkpoint_first: bool) { + let mut f = Fixture::new().await; + let mut cell = f.cell(4).await; + let (_, frames, assigned) = f.append(&mut cell, 2); + let prepared = f + .directory + .prepare_node_bundle(&f.node, &frames, &[assigned], NOW) + .await + .unwrap(); + let (node, mut proofs) = f + .directory + .select_node_bundle(&f.node, &prepared, &f.lease, Limits::default(), NOW) + .await + .unwrap(); + let old = proofs.pop().unwrap(); + f.node = node; + // The writer advances selection while the older materialization is pending. + let (_, frames, assigned) = f.append(&mut cell, 3); + let prepared = f + .directory + .prepare_node_bundle(&f.node, &frames, &[assigned], NOW) + .await + .unwrap(); + let (latest, mut proofs) = f + .directory + .select_node_bundle(&f.node, &prepared, &f.lease, Limits::default(), NOW) + .await + .unwrap(); + let mut publisher = f.publisher(&cell); + let prefix = publisher.materialize_bundle(&old).await.unwrap(); + assert_eq!(prefix.commit_sequence, 2); + let current = cell + .authority + .load(cell.control.value().cell) + .await + .unwrap() + .unwrap(); + let between = f + .directory + .load_bundle_coverage(&cell.authority, ¤t, Limits::default()) + .await + .unwrap(); + assert_eq!(between.commit_sequence(), 3); + let pin = cell.control.value().bundle_binding.unwrap(); + let (node, proof) = if checkpoint_first { + let checkpoint = f + .directory + .checkpoint_bundle_cell(&latest, &cell.authority, pin, Limits::default(), NOW) + .await + .unwrap(); + let current = cell + .authority + .load(cell.control.value().cell) + .await + .unwrap() + .unwrap(); + let proof = f + .directory + .load_bundle_coverage(&cell.authority, ¤t, Limits::default()) + .await + .unwrap(); + assert_eq!(proof.base().unwrap(), prefix); + assert_eq!(proof.commit_sequence(), 3); + assert_eq!(proof.locator_count(), frames.len()); + (checkpoint, proof) + } else { + // Origin lookup works between the root CAS and the catalog checkpoint. + assert_eq!(between.binding(), proofs.pop().unwrap().binding()); + (latest, between) + }; + let final_root = publisher.materialize_bundle(&proof).await.unwrap(); + assert_eq!(final_root.commit_sequence, 3); + assert_eq!( + publisher.materialize_bundle(&proof).await.unwrap(), + final_root + ); + let checkpoint = f + .directory + .checkpoint_bundle_cell(&node, &cell.authority, pin, Limits::default(), NOW) + .await + .unwrap(); + let current = cell + .authority + .load(cell.control.value().cell) + .await + .unwrap() + .unwrap(); + let final_proof = f + .directory + .load_bundle_coverage(&cell.authority, ¤t, Limits::default()) + .await + .unwrap(); + assert_eq!(final_proof.locator_count(), 0); + assert_eq!(final_proof.base().unwrap(), final_root); + assert_eq!( + checkpoint + .advertisement() + .bundle_head() + .unwrap() + .selected_through(), + assigned.ticket().last_sequence() + ); +} + +#[tokio::test] +async fn checkpoint_releases_locators_and_the_next_range_continues_exactly() { + let mut f = Fixture::new().await; + let mut cell = f.cell(4).await; + let (_, frames, assigned) = f.append(&mut cell, 2); + let prepared = f + .directory + .prepare_node_bundle(&f.node, &frames, &[assigned], NOW) + .await + .unwrap(); + let (selected, mut proofs) = f + .directory + .select_node_bundle(&f.node, &prepared, &f.lease, Limits::default(), NOW) + .await + .unwrap(); + let pin = cell.control.value().bundle_binding.unwrap(); + assert!(matches!( + f.directory + .checkpoint_bundle_cell(&selected, &cell.authority, pin, Limits::default(), NOW) + .await, + Err(Error::PendingPublication) + )); + let proof = proofs.pop().unwrap(); + let mut publisher = f.publisher(&cell); + let materialized = publisher.materialize_bundle(&proof).await.unwrap(); + cell.control = cell + .authority + .load(cell.control.value().cell) + .await + .unwrap() + .unwrap(); + f.node = f + .directory + .checkpoint_bundle_cell(&selected, &cell.authority, pin, Limits::default(), NOW) + .await + .unwrap(); + let checkpoint = f + .directory + .load_bundle_coverage(&cell.authority, &cell.control, Limits::default()) + .await + .unwrap(); + assert_eq!(checkpoint.base().unwrap(), materialized); + assert_eq!(checkpoint.locator_count(), 0); + let (_, frames, assigned) = f.append(&mut cell, 3); + let prepared = f + .directory + .prepare_node_bundle(&f.node, &frames, &[assigned], NOW) + .await + .unwrap(); + let (_, proofs) = f + .directory + .select_node_bundle(&f.node, &prepared, &f.lease, Limits::default(), NOW) + .await + .unwrap(); + assert_eq!(proofs[0].base().unwrap(), materialized); + assert_eq!(proofs[0].commit_sequence(), 3); + assert_eq!(proofs[0].locator_count(), frames.len()); +} + +#[tokio::test] +async fn quiet_binding_must_close_and_checkpoint_before_session_withdrawal() { + let mut f = Fixture::new().await; + let cell = f.cell(4).await; + assert!(matches!( + f.directory.withdraw(&f.node, NOW).await, + Err(Error::PendingPublication) + )); + let pin = cell.control.value().bundle_binding.unwrap(); + let issued = f + .gate + .close_cell_issuance( + Fixture::scope(&cell), + cell.control.value().ltx_root().unwrap(), + ) + .unwrap(); + assert_eq!(issued.last_node_sequence(), 0); + let closing = f + .directory + .begin_bundle_close(&f.node, pin, issued, NOW) + .await + .unwrap(); + let closed = f + .directory + .finish_bundle_close(&closing, pin, NOW) + .await + .unwrap(); + let checkpoint = f + .directory + .checkpoint_bundle_cell(&closed, &cell.authority, pin, Limits::default(), NOW) + .await + .unwrap(); + f.directory.withdraw(&checkpoint, NOW).await.unwrap(); + assert!(f.directory.is_withdrawn(pin.session).await.unwrap()); +} + +#[tokio::test] +async fn selected_suffix_survives_original_database_loss_and_node_fencing() { + let mut f = Fixture::new().await; + let mut cell = f.cell(4).await; + let (cuts, frames, assigned) = f.append(&mut cell, 2); + let position = cuts.position; + let prepared = f + .directory + .prepare_node_bundle(&f.node, &frames, &[assigned], NOW) + .await + .unwrap(); + let (selected, _) = f + .directory + .select_node_bundle(&f.node, &prepared, &f.lease, Limits::default(), NOW) + .await + .unwrap(); + drop(cell.db); + std::fs::remove_dir_all(f.scratch.path()).unwrap(); + let old = NOW + 30_000 + crate::node::STALE_ADVERTISEMENT_RETENTION_MS; + // The node-record scanner must see only records, never coverage data objects. + assert_eq!(f.directory.collect_stale(old, 1).await.unwrap(), 1); + assert!( + f.directory + .is_retired(SessionId::from_bytes([1; 16])) + .await + .unwrap() + ); + assert!( + !f.directory + .is_withdrawn(SessionId::from_bytes([1; 16])) + .await + .unwrap() + ); + let record = f + .directory + .layout + .store() + .get_with_etag_bounded(&f.layout.node_path(&[1; 16]), crate::node::MAX_NODE_BYTES) + .await + .unwrap() + .0; + let crate::node::directory::NodeRecord::Tombstone(record) = + crate::node::directory::NodeRecord::decode_canonical(&record).unwrap() + else { + panic!("original boot must be fenced"); + }; + assert_eq!(record.bundle, selected.advertisement().bundle_head()); + let reopened = f + .directory + .load_bundle_coverage(&cell.authority, &cell.control, Limits::default()) + .await + .unwrap(); + let overlay = reopened + .recovery_overlay(&f.layout, Limits::default()) + .await + .unwrap(); + let prepared = cell + .replica + .prepare_recovered_overlay(&overlay, 1) + .await + .unwrap(); + assert_eq!(prepared.root().position, position); + let recovery = tempfile::tempdir().unwrap(); + let file = recovery.path().join("restored.sqlite"); + cell.replica + .open_root(&prepared.root()) + .await + .unwrap() + .restore(&file) + .await + .unwrap(); + let db = rusqlite::Connection::open(file).unwrap(); + let result: String = db + .query_row( + "SELECT result FROM outcomes WHERE request='request-2'", + [], + |row| row.get(0), + ) + .unwrap(); + assert_eq!(result, "result-2"); +} + +#[tokio::test] +async fn materializer_failure_preserves_selected_proof_and_lagging_root() { + let mut f = Fixture::new().await; + let mut cell = f.cell(4).await; + let (_, frames, assigned) = f.append(&mut cell, 2); + let prepared = f + .directory + .prepare_node_bundle(&f.node, &frames, &[assigned], NOW) + .await + .unwrap(); + let (selected, proofs) = f + .directory + .select_node_bundle(&f.node, &prepared, &f.lease, Limits::default(), NOW) + .await + .unwrap(); + let path = f + .layout + .node_coverage_bundle_path(&[1; 16], EPOCH, prepared.head.digest.as_bytes()); + f.count.block_body_reads_for(&path); + let mut publisher = f.publisher(&cell); + assert!(publisher.materialize_bundle(&proofs[0]).await.is_err()); + assert_eq!( + cell.authority + .load(cell.control.value().cell) + .await + .unwrap() + .unwrap() + .value() + .root, + cell.control.value().root + ); + assert_eq!( + f.directory + .load(SessionId::from_bytes([1; 16]), NOW) + .await + .unwrap() + .unwrap() + .advertisement() + .bundle_head(), + selected.advertisement().bundle_head() + ); + f.count.unblock_body_reads_for(&path); + assert_eq!( + publisher + .materialize_bundle(&proofs[0]) + .await + .unwrap() + .commit_sequence, + 2 + ); +} + +#[tokio::test] +async fn untracked_legacy_issuance_cannot_close_a_cell() { + let mut f = Fixture::new().await; + let cell = f.cell(4).await; + f.gate.issue(1).unwrap(); + assert!(matches!( + f.gate.close_cell_issuance( + Fixture::scope(&cell), + cell.control.value().ltx_root().unwrap() + ), + Err(Error::Fenced) + )); +} diff --git a/crates/cellule-runtime/src/node/bundle/tests/mod.rs b/crates/cellule-runtime/src/node/bundle/tests/mod.rs new file mode 100644 index 00000000..887b9887 --- /dev/null +++ b/crates/cellule-runtime/src/node/bundle/tests/mod.rs @@ -0,0 +1,586 @@ +use super::*; +use crate::control::authority::{CellAuthority, VersionedControl}; +use crate::control::{ControlState, Owner, RootRef, Transition}; +use crate::identity::{ApplicationId, CellId, IncarnationId, NodeId}; +use crate::node::lease::NodeLeaseGuard; +use crate::node::log::{AssignedCommitRange, CellLogScope, DurabilityGate, DurabilitySource}; +use crate::node::{ + NodeAdvertisement, NodeCapacity, NodeDirectory, NodeFailureDomain, VersionedNodeAdvertisement, +}; +use cellule_ltx::{CellReplica, CellStorageLayout, Db, Limits}; +use cellule_store::{Store, test_support::CountingObjectStore}; +use ed25519_dalek::{Signer, SigningKey}; +use object_store::{memory::InMemory, path::Path}; +use std::sync::Arc; + +const NOW: i64 = 1_000_000; +const EPOCH: u64 = 2; +mod faults; +mod lifecycle; +mod ranges; +struct Fixture { + count: Arc, + layout: CellStorageLayout, + directory: NodeDirectory, + node: VersionedNodeAdvertisement, + lease: NodeLeaseGuard, + gate: DurabilityGate, + scratch: tempfile::TempDir, +} +struct Cell { + db: Db, + replica: CellReplica, + authority: CellAuthority, + control: VersionedControl, +} +impl Fixture { + async fn new() -> Self { + Self::with_store(Arc::new(InMemory::new())).await + } + async fn with_store(store: Arc) -> Self { + let count = Arc::new(CountingObjectStore::new(store)); + let layout = CellStorageLayout::new( + Store::new(count.clone()), + Path::from("bundle-test"), + [9; 16], + ); + let directory = NodeDirectory::new( + layout.clone(), + Digest::from_bytes([6; 32]), + Digest::from_bytes([7; 32]), + Digest::from_bytes([8; 32]), + ); + let key = SigningKey::from_bytes(&[10; 32]); + let advertisement = NodeAdvertisement::sign( + NodeId::from_bytes([1; 16]), + SessionId::from_bytes([1; 16]), + "https://bundle.internal:8081".into(), + Digest::from_bytes([6; 32]), + Digest::from_bytes([11; 32]), + Digest::from_bytes([7; 32]), + Digest::from_bytes([8; 32]), + &key, + 1, + NOW, + NOW + 30_000, + vec![Digest::from_bytes([12; 32])], + vec![1], + NodeFailureDomain::new(None, None).unwrap(), + NodeCapacity { + free_memory_bytes: 1 << 30, + free_disk_bytes: 1 << 30, + follower_free_bytes: 1 << 30, + follower_retained_bytes: 0, + job_credits: 8, + log_protocol: 2, + }, + ) + .unwrap(); + let node = directory.create(advertisement, NOW).await.unwrap(); + let node = directory + .initialize_bundle_lane(&node, EPOCH, NOW) + .await + .unwrap(); + let lease = NodeLeaseGuard::new(NOW, NOW + 30_000).unwrap(); + let gate = DurabilityGate::new( + SessionId::from_bytes([1; 16]), + NodeId::from_bytes([1; 16]), + EPOCH, + [NodeId::from_bytes([2; 16]), NodeId::from_bytes([3; 16])], + ) + .unwrap(); + Self { + count, + layout, + directory, + node, + lease, + gate, + scratch: tempfile::tempdir().unwrap(), + } + } + async fn cell(&mut self, byte: u8) -> Cell { + self.cell_for_application(byte, [9; 16]).await + } + async fn cell_for_application(&mut self, byte: u8, application: [u8; 16]) -> Cell { + let layout = self.layout.for_application(application); + let cell = CellId::from_bytes([byte; 32]); + let incarnation = IncarnationId::from_bytes([byte + 10; 16]); + let mut db = Db::open( + &self.scratch.path().join(format!( + "{byte}-{}.sqlite", + crate::identity::encode_hex(&application) + )), + Limits::default(), + ) + .unwrap(); + db.transaction(|tx| tx.execute_batch("CREATE TABLE outcomes(request TEXT PRIMARY KEY, result TEXT); INSERT INTO outcomes VALUES ('seed','original')")).unwrap(); + let cuts = db.capture().unwrap(); + let replica = CellReplica::new( + layout.clone(), + *cell.as_bytes(), + *incarnation.as_bytes(), + Limits::default(), + ) + .unwrap(); + let prepared = replica.prepare(None, &cuts, 1, 1).await.unwrap(); + let mut control = Control::initial( + cell, + incarnation, + Owner { + session: SessionId::from_bytes([1; 16]), + endpoint: "https://bundle.internal:8081".into(), + }, + Digest::from_bytes([12; 32]), + 1, + ) + .unwrap(); + control.state = ControlState::Serving; + control.root = Some(RootRef::from_ltx(cell, incarnation, prepared.root()).unwrap()); + layout + .store() + .create_strict( + &layout.control_path(cell.as_bytes()), + Bytes::from(control.encode().unwrap()), + ) + .await + .unwrap(); + let authority = CellAuthority::new(layout); + authority.retain_root_lineage(&prepared).await.unwrap(); + let observed = authority.load(cell).await.unwrap().unwrap(); + let (node, control) = self + .directory + .bind_bundle_cell(&self.node, &authority, &observed, NOW) + .await + .unwrap(); + self.node = node; + Cell { + db, + replica, + authority, + control, + } + } + fn append( + &self, + cell: &mut Cell, + commit: u64, + ) -> ( + cellule_ltx::CaptureBatch, + Vec, + AssignedCommitRange, + ) { + cell.db + .transaction(|tx| { + tx.execute( + "INSERT INTO outcomes VALUES (?1,?2)", + [format!("request-{commit}"), format!("result-{commit}")], + ) + }) + .unwrap(); + let cuts = cell.db.capture().unwrap(); + let ticket = self.gate.preview(cuts.segments.len() as u64).unwrap(); + let frames = cuts + .segments + .iter() + .enumerate() + .map(|(offset, segment)| { + cellule_ltx::encode_node_frame( + cellule_ltx::NodeFrameScope { + leader_session: [1; 16], + log_epoch: EPOCH, + node_sequence: ticket.first_sequence() + offset as u64, + application: *cell.authority.layout().application_id(), + cell: *cell.control.value().cell.as_bytes(), + incarnation: *cell.control.value().incarnation.as_bytes(), + cell_epoch: cell.control.value().epoch, + commit_sequence: commit, + }, + segment.info().clone(), + Bytes::from(std::fs::read(segment.path()).unwrap()), + Limits::default(), + ) + .unwrap() + }) + .collect::>(); + let assignment = self.gate.commit_frames(ticket, &frames).unwrap().unwrap(); + (cuts, frames, assignment) + } + fn scope(cell: &Cell) -> CellLogScope { + CellLogScope { + application: ApplicationId::from_bytes(*cell.authority.layout().application_id()), + cell: cell.control.value().cell, + incarnation: cell.control.value().incarnation, + cell_epoch: cell.control.value().epoch, + } + } + fn publisher(&self, cell: &Cell) -> crate::publication::CellPublisher { + crate::publication::CellPublisher::new( + cell.replica.clone(), + cell.authority.clone(), + cell.control.clone(), + self.scratch.path().to_owned(), + ) + .with_node_lease(self.lease.clone()) + } +} + +#[tokio::test] +async fn one_selection_covers_two_cells_while_roots_lag_and_materializes_identical_bytes() { + let mut f = Fixture::new().await; + let mut a = f.cell(4).await; + let mut b = f.cell(5).await; + let (a_cuts, mut frames, a_range) = f.append(&mut a, 2); + let (b_cuts, b_frames, b_range) = f.append(&mut b, 2); + frames.extend(b_frames); + let before_a = a.control.value().encode().unwrap(); + let before_b = b.control.value().encode().unwrap(); + f.count.reset(); + let prepared = f + .directory + .prepare_node_bundle(&f.node, &frames, &[a_range, b_range], NOW) + .await + .unwrap(); + assert_eq!(f.count.put_requests(), 1); + let (node, proofs) = f + .directory + .select_node_bundle(&f.node, &prepared, &f.lease, Limits::default(), NOW) + .await + .unwrap(); + assert_eq!( + f.count.put_requests(), + 2, + "one immutable upload and one shared selector CAS" + ); + assert_eq!(proofs.len(), 2); + assert_eq!( + a.authority + .load(a.control.value().cell) + .await + .unwrap() + .unwrap() + .value() + .encode() + .unwrap(), + before_a + ); + assert_eq!( + b.authority + .load(b.control.value().cell) + .await + .unwrap() + .unwrap() + .value() + .encode() + .unwrap(), + before_b + ); + assert_eq!( + node.advertisement() + .bundle_head() + .unwrap() + .selected_through(), + frames.len() as u64 + ); + for (cell, cuts) in [(&a, &a_cuts), (&b, &b_cuts)] { + let proof = proofs + .iter() + .find(|proof| proof.binding() == cell.control.value().bundle_binding.unwrap()) + .unwrap(); + assert_eq!(proof.locator_count(), cuts.segments.len()); + let expected = cell + .replica + .prepare(cell.control.value().ltx_root().as_ref(), cuts, 2, 1) + .await + .unwrap(); + let actual = f.publisher(cell).materialize_bundle(proof).await.unwrap(); + assert_eq!(actual.position, cuts.position); + let expected_path = f.scratch.path().join(format!( + "expected-{}.sqlite", + cell.control.value().cell.as_bytes()[0] + )); + let actual_path = f.scratch.path().join(format!( + "actual-{}.sqlite", + cell.control.value().cell.as_bytes()[0] + )); + cell.replica + .open_root(&expected.root()) + .await + .unwrap() + .restore(&expected_path) + .await + .unwrap(); + cell.replica + .open_root(&actual) + .await + .unwrap() + .restore(&actual_path) + .await + .unwrap(); + assert_eq!( + std::fs::read(actual_path).unwrap(), + std::fs::read(expected_path).unwrap() + ); + } +} + +#[tokio::test] +async fn closure_drains_prior_fleet_ack_rejects_old_cas_and_blocks_transfer_until_materialized() { + let mut f = Fixture::new().await; + let mut a = f.cell(4).await; + let mut b = f.cell(5).await; + let (_, frames, assignment) = f.append(&mut a, 2); + for member in [2, 3] { + f.gate + .acknowledge( + NodeId::from_bytes([member; 16]), + assignment.ticket().last_sequence(), + ) + .unwrap(); + } + f.gate.activate_fleet().unwrap(); + assert_eq!( + f.gate.prove(assignment.ticket()).await.unwrap().source(), + DurabilitySource::Fleet + ); + assert_eq!( + f.node + .advertisement() + .bundle_head() + .unwrap() + .selected_through(), + 0 + ); + let old = f + .directory + .prepare_node_bundle(&f.node, &frames, &[assignment], NOW) + .await + .unwrap(); + let issued = f + .gate + .close_cell_issuance(Fixture::scope(&a), a.control.value().ltx_root().unwrap()) + .unwrap(); + assert_eq!(issued.commit_sequence(), 2); + let pin = a.control.value().bundle_binding.unwrap(); + let closing = f + .directory + .begin_bundle_close(&f.node, pin, issued, NOW) + .await + .unwrap(); + assert!( + f.directory + .select_node_bundle(&closing, &old, &f.lease, Limits::default(), NOW) + .await + .is_err() + ); + assert!(matches!( + f.directory.finish_bundle_close(&closing, pin, NOW).await, + Err(Error::PendingPublication) + )); + let successor = a + .control + .value() + .takeover(Owner { + session: SessionId::from_bytes([6; 16]), + endpoint: "https://successor.internal:8081".into(), + }) + .unwrap(); + assert!(matches!( + a.authority + .transition(&a.control, successor, Transition::Takeover) + .await, + Err(Error::PendingPublication) + )); + let prepared = f + .directory + .prepare_node_bundle(&closing, &frames, &[assignment], NOW) + .await + .unwrap(); + let (selected, proofs) = f + .directory + .select_node_bundle(&closing, &prepared, &f.lease, Limits::default(), NOW) + .await + .unwrap(); + let closed = f + .directory + .finish_bundle_close(&selected, pin, NOW) + .await + .unwrap(); + assert!(matches!( + a.authority + .transition( + &a.control, + a.control.value().release().unwrap(), + Transition::Release + ) + .await, + Err(Error::PendingPublication) + )); + let mut publisher = f.publisher(&a); + publisher.materialize_bundle(&proofs[0]).await.unwrap(); + let materialized = publisher.control().clone(); + assert!(matches!( + a.authority + .transition( + &materialized, + materialized.value().release().unwrap(), + Transition::Release + ) + .await, + Err(Error::PendingPublication) + )); + // Checkpoint before clearing the Cell pin; otherwise the immutable catalog + // would retain locators that the departed writer can no longer release. + let closed = f + .directory + .checkpoint_bundle_cell(&closed, &a.authority, pin, Limits::default(), NOW) + .await + .unwrap(); + let released = a + .authority + .transition( + &materialized, + materialized.value().release().unwrap(), + Transition::Release, + ) + .await + .unwrap(); + assert!(released.value().bundle_binding.is_none()); + assert!( + f.directory + .bind_bundle_cell(&closed, &a.authority, &a.control, NOW) + .await + .is_err() + ); + let (_, sibling_frames, sibling_assignment) = f.append(&mut b, 2); + let prepared = f + .directory + .prepare_node_bundle(&closed, &sibling_frames, &[sibling_assignment], NOW) + .await + .unwrap(); + f.directory + .select_node_bundle(&closed, &prepared, &f.lease, Limits::default(), NOW) + .await + .unwrap(); + let ticket = f.gate.preview(1).unwrap(); + let late = frames[0] + .clone() + .with_node_sequence(ticket.first_sequence()) + .unwrap(); + assert!(matches!( + f.gate.commit_frames(ticket, &[late]), + Err(Error::Fenced) + )); + assert_eq!( + f.gate.issued_through(), + sibling_assignment.ticket().last_sequence() + ); +} + +#[tokio::test] +async fn missing_and_unassigned_rows_never_mint_proof_or_advance_head() { + let mut f = Fixture::new().await; + let mut cell = f.cell(4).await; + let (_, frames, assignment) = f.append(&mut cell, 2); + assert!( + f.directory + .prepare_node_bundle(&f.node, &frames, &[], NOW) + .await + .is_err() + ); + assert!( + f.directory + .prepare_node_bundle(&f.node, &[], &[assignment], NOW) + .await + .is_err() + ); + let mut wrong = frames.clone(); + wrong[0] = wrong[0].clone().with_node_sequence(2).unwrap(); + assert!( + f.directory + .prepare_node_bundle(&f.node, &wrong, &[assignment], NOW) + .await + .is_err() + ); + let prepared = f + .directory + .prepare_node_bundle(&f.node, &frames, &[assignment], NOW) + .await + .unwrap(); + let path = f + .layout + .node_coverage_bundle_path(&[1; 16], EPOCH, prepared.head.digest.as_bytes()); + f.layout.store().delete(&path).await.unwrap(); + assert!( + f.directory + .select_node_bundle(&f.node, &prepared, &f.lease, Limits::default(), NOW) + .await + .is_err() + ); + let node = f + .directory + .load(SessionId::from_bytes([1; 16]), NOW) + .await + .unwrap() + .unwrap(); + assert_eq!( + node.advertisement().bundle_head(), + f.node.advertisement().bundle_head() + ); + assert_eq!( + cell.authority + .load(cell.control.value().cell) + .await + .unwrap() + .unwrap() + .value() + .root, + cell.control.value().root + ); +} + +#[tokio::test] +async fn lease_loss_and_heartbeat_rebase_preserve_bundle_authority() { + let mut f = Fixture::new().await; + let mut cell = f.cell(4).await; + let (_, frames, assignment) = f.append(&mut cell, 2); + let prepared = f + .directory + .prepare_node_bundle(&f.node, &frames, &[assignment], NOW) + .await + .unwrap(); + let key = SigningKey::from_bytes(&[10; 32]); + let mut next = f.node.advertisement().clone(); + next.issued_at_ms += 1_000; + next.expires_at_ms += 1_000; + next.progress += 1; + next.signature = key.sign(&next.signing_bytes().unwrap()).to_bytes(); + let refreshed = f + .directory + .refresh(&f.node, next, NOW + 1_000) + .await + .unwrap(); + assert_eq!( + refreshed.advertisement().bundle_head(), + f.node.advertisement().bundle_head() + ); + let (selected, _) = f + .directory + .select_node_bundle(&f.node, &prepared, &f.lease, Limits::default(), NOW + 1_000) + .await + .unwrap(); + assert_eq!(selected.advertisement().issued_at_ms(), NOW + 1_000); + f.lease.fence(); + assert!(matches!( + f.directory + .select_node_bundle( + &selected, + &prepared, + &f.lease, + Limits::default(), + NOW + 1_000 + ) + .await, + Err(Error::Fenced) + )); +} diff --git a/crates/cellule-runtime/src/node/bundle/tests/ranges.rs b/crates/cellule-runtime/src/node/bundle/tests/ranges.rs new file mode 100644 index 00000000..6f947137 --- /dev/null +++ b/crates/cellule-runtime/src/node/bundle/tests/ranges.rs @@ -0,0 +1,313 @@ +use super::*; +use object_store::ObjectStoreExt as _; + +#[tokio::test] +async fn missing_base_chunk_cannot_mint_a_selected_reconstruction_proof() { + let mut f = Fixture::new().await; + let mut cell = f.cell(4).await; + let base = cell.control.value().ltx_root().unwrap(); + let objects = cell.replica.reachable_objects(&base).await.unwrap(); + let dependency = objects + .into_iter() + .find(|object| object.kind != cellule_ltx::CellObjectKind::Root) + .unwrap(); + let (_, frames, assigned) = f.append(&mut cell, 2); + let prepared = f + .directory + .prepare_node_bundle(&f.node, &frames, &[assigned], NOW) + .await + .unwrap(); + let path = f.layout.incarnation_object_path( + cell.control.value().cell.as_bytes(), + cell.control.value().incarnation.as_bytes(), + &dependency.digest, + dependency.kind, + ); + f.layout.store().delete(&path).await.unwrap(); + assert!( + f.directory + .select_node_bundle(&f.node, &prepared, &f.lease, Limits::default(), NOW) + .await + .is_err() + ); + assert_eq!( + f.directory + .load(SessionId::from_bytes([1; 16]), NOW) + .await + .unwrap() + .unwrap() + .advertisement() + .bundle_head(), + f.node.advertisement().bundle_head() + ); +} + +#[tokio::test] +async fn same_cell_ids_in_two_applications_keep_separate_bases_and_proofs() { + let mut f = Fixture::new().await; + let mut a = f.cell(4).await; + let mut b = f.cell_for_application(4, [10; 16]).await; + let (_, mut frames, first) = f.append(&mut a, 2); + let (_, extra, second) = f.append(&mut b, 2); + frames.extend(extra); + let prepared = f + .directory + .prepare_node_bundle(&f.node, &frames, &[first, second], NOW) + .await + .unwrap(); + let (_, proofs) = f + .directory + .select_node_bundle(&f.node, &prepared, &f.lease, Limits::default(), NOW) + .await + .unwrap(); + assert_eq!(proofs.len(), 2); + for cell in [a, b] { + let proof = proofs + .iter() + .find(|proof| Some(proof.binding()) == cell.control.value().bundle_binding) + .unwrap(); + let mut publisher = f.publisher(&cell); + assert_eq!( + publisher + .materialize_bundle(proof) + .await + .unwrap() + .commit_sequence, + 2 + ); + } +} + +#[tokio::test] +async fn forward_native_issuance_cannot_hide_a_missing_object_only_command() { + let mut f = Fixture::new().await; + let mut cell = f.cell(4).await; + let (_, frames, assigned) = f.append(&mut cell, 2); + let prepared = f + .directory + .prepare_node_bundle(&f.node, &frames, &[assigned], NOW) + .await + .unwrap(); + f.node = f + .directory + .select_node_bundle(&f.node, &prepared, &f.lease, Limits::default(), NOW) + .await + .unwrap() + .0; + cell.db + .transaction(|tx| tx.execute_batch("INSERT INTO outcomes VALUES ('request-3', 'result-3')")) + .unwrap(); + let _object_only = cell.db.capture().unwrap(); + let (_, next, assigned) = f.append(&mut cell, 4); + assert!( + f.directory + .prepare_node_bundle(&f.node, &next, &[assigned], NOW) + .await + .is_err() + ); + assert_eq!( + f.directory + .load_bundle_coverage(&cell.authority, &cell.control, Limits::default()) + .await + .unwrap() + .commit_sequence(), + 2 + ); +} + +#[tokio::test] +async fn a_native_capture_group_cannot_be_selected_at_a_partial_endpoint() { + let mut f = Fixture::new().await; + let mut cell = f.cell(4).await; + let mut cuts = Vec::new(); + for commit in [2, 3] { + cell.db + .transaction(|tx| { + tx.execute( + "INSERT INTO outcomes VALUES (?1,?2)", + [format!("request-{commit}"), format!("result-{commit}")], + ) + }) + .unwrap(); + cuts.push(cell.db.capture().unwrap()); + } + let segments: Vec<_> = cuts.iter().flat_map(|cut| &cut.segments).collect(); + assert!(segments.len() > 1); + let ticket = f.gate.preview(segments.len() as u64).unwrap(); + let frames: Vec<_> = segments + .iter() + .enumerate() + .map(|(offset, segment)| { + cellule_ltx::encode_node_frame_range( + cellule_ltx::NodeFrameScope { + leader_session: [1; 16], + log_epoch: EPOCH, + node_sequence: ticket.first_sequence() + offset as u64, + application: [9; 16], + cell: *cell.control.value().cell.as_bytes(), + incarnation: *cell.control.value().incarnation.as_bytes(), + cell_epoch: cell.control.value().epoch, + commit_sequence: 3, + }, + 2, + segment.info().clone(), + Bytes::from(std::fs::read(segment.path()).unwrap()), + Limits::default(), + ) + .unwrap() + }) + .collect(); + let assigned = f.gate.commit_frames(ticket, &frames).unwrap().unwrap(); + assert!( + f.directory + .prepare_node_bundle(&f.node, &frames[..frames.len() - 1], &[assigned], NOW) + .await + .is_err() + ); + let prepared = f + .directory + .prepare_node_bundle(&f.node, &frames, &[assigned], NOW) + .await + .unwrap(); + let (_, proofs) = f + .directory + .select_node_bundle(&f.node, &prepared, &f.lease, Limits::default(), NOW) + .await + .unwrap(); + assert_eq!(proofs[0].commit_sequence(), 3); + assert_eq!(proofs[0].position(), cuts.last().unwrap().position); + let mut partial = proofs[0].base().unwrap(); + partial.commit_sequence = 3; + partial.position = frames[0].segment().position(); + assert!(checkpoint_prefix(&proofs[0].binding, &frames, &partial).is_err()); +} + +#[tokio::test] +async fn corrupt_origin_dependency_and_overlapping_ranges_fail_closed() { + let mut f = Fixture::new().await; + let mut cell = f.cell(4).await; + let (_, frames, assigned) = f.append(&mut cell, 2); + let mut duplicate = frames.clone(); + duplicate.extend(frames.clone()); + assert!( + f.directory + .prepare_node_bundle(&f.node, &duplicate, &[assigned, assigned], NOW) + .await + .is_err() + ); + let prepared = f + .directory + .prepare_node_bundle(&f.node, &frames, &[assigned], NOW) + .await + .unwrap(); + let (selected, _) = f + .directory + .select_node_bundle(&f.node, &prepared, &f.lease, Limits::default(), NOW) + .await + .unwrap(); + f.node = selected; + let (_, next, assigned) = f.append(&mut cell, 3); + let proposal = f + .directory + .prepare_node_bundle(&f.node, &next, &[assigned], NOW) + .await + .unwrap(); + let path = f + .layout + .node_coverage_bundle_path(&[1; 16], EPOCH, prepared.head.digest.as_bytes()); + let mut corrupt = prepared.body.to_vec(); + let last = corrupt.last_mut().unwrap(); + *last ^= 1; + f.layout + .store() + .inner() + .put(&path, Bytes::from(corrupt).into()) + .await + .unwrap(); + assert!( + f.directory + .select_node_bundle(&f.node, &proposal, &f.lease, Limits::default(), NOW) + .await + .is_err() + ); + assert_eq!( + f.directory + .load(SessionId::from_bytes([1; 16]), NOW) + .await + .unwrap() + .unwrap() + .advertisement() + .bundle_head(), + f.node.advertisement().bundle_head() + ); + assert!( + f.directory + .load_bundle_coverage(&cell.authority, &cell.control, Limits::default()) + .await + .is_err() + ); +} + +#[tokio::test] +async fn locator_pressure_refuses_selection_without_dropping_the_last_proof() { + let mut f = Fixture::new().await; + let mut cell = f.cell(4).await; + for commit in 2..=(MAX_LOCATORS as u64 + 1) { + let (_, frames, assigned) = f.append(&mut cell, commit); + let prepared = f + .directory + .prepare_node_bundle(&f.node, &frames, &[assigned], NOW) + .await + .unwrap(); + f.node = f + .directory + .select_node_bundle(&f.node, &prepared, &f.lease, Limits::default(), NOW) + .await + .unwrap() + .0; + } + let last = f + .directory + .load_bundle_coverage(&cell.authority, &cell.control, Limits::default()) + .await + .unwrap(); + assert_eq!(last.locator_count(), MAX_LOCATORS); + let (_, frames, assigned) = f.append(&mut cell, MAX_LOCATORS as u64 + 2); + assert!(matches!( + f.directory + .prepare_node_bundle(&f.node, &frames, &[assigned], NOW) + .await, + Err(Error::PendingPublication) + )); + assert_eq!(last.commit_sequence(), MAX_LOCATORS as u64 + 1); + assert_eq!( + f.directory + .load_bundle_coverage(&cell.authority, &cell.control, Limits::default()) + .await + .unwrap() + .commit_sequence(), + last.commit_sequence() + ); +} + +#[tokio::test] +async fn version_cutover_rejects_erased_and_unknown_authority_fields() { + let mut f = Fixture::new().await; + let cell = f.cell(4).await; + let mut control: serde_json::Value = + serde_json::from_slice(&cell.control.value().encode().unwrap()).unwrap(); + assert_eq!(control["version"], 2); + control["version"] = 1.into(); + assert!(Control::decode(&serde_json::to_vec(&control).unwrap()).is_err()); + control["version"] = 2.into(); + control.as_object_mut().unwrap().remove("bundle_binding"); + assert!(Control::decode(&serde_json::to_vec(&control).unwrap()).is_err()); + let bytes = f.node.advertisement().encode().unwrap(); + let mut node: serde_json::Value = serde_json::from_slice(&bytes).unwrap(); + assert_eq!(node["version"], 2); + node["version"] = 1.into(); + assert!(NodeAdvertisement::decode_canonical(&serde_json::to_vec(&node).unwrap()).is_err()); + node["version"] = 2.into(); + node["bundle"]["unrecognized"] = 1.into(); + assert!(NodeAdvertisement::decode_canonical(&serde_json::to_vec(&node).unwrap()).is_err()); +} diff --git a/crates/cellule-runtime/src/node/directory/advertisement.rs b/crates/cellule-runtime/src/node/directory/advertisement.rs index 8d23ff87..fc982fe6 100644 --- a/crates/cellule-runtime/src/node/directory/advertisement.rs +++ b/crates/cellule-runtime/src/node/directory/advertisement.rs @@ -457,6 +457,7 @@ impl NodeDirectory { now_ms, None, advertisement.log.clone(), + advertisement.bundle, )?; let encoded = tombstone.encode()?; match self @@ -496,6 +497,12 @@ impl NodeDirectory { "node log must be sealed before session withdrawal", )); } + crate::node::bundle::store::ensure_session_drained( + &self.layout, + observed.advertisement.session, + observed.advertisement.bundle, + ) + .await?; let path = self .layout .node_path(observed.advertisement.session.as_bytes()); @@ -506,6 +513,7 @@ impl NodeDirectory { now_ms, None, None, + observed.advertisement.bundle, )?; let result = match self .layout @@ -529,6 +537,12 @@ impl NodeDirectory { "node log must be sealed before session withdrawal", )) } else { + crate::node::bundle::store::ensure_session_drained( + &self.layout, + current.session, + current.bundle, + ) + .await?; Ok(()) } } @@ -668,6 +682,7 @@ impl NodeDirectory { .checked_add(1) .ok_or(Error::Node("node session generation overflow"))?; candidate.log.clone_from(&base.advertisement.log); + candidate.bundle = base.advertisement.bundle; self.validate(&candidate, now_ms)?; validate_successor(&base.advertisement, &candidate)?; // The validated candidate remains immutable through its CAS body. diff --git a/crates/cellule-runtime/src/node/directory/closure/mod.rs b/crates/cellule-runtime/src/node/directory/closure/mod.rs index 171f5c3c..4cdbb0bb 100644 --- a/crates/cellule-runtime/src/node/directory/closure/mod.rs +++ b/crates/cellule-runtime/src/node/directory/closure/mod.rs @@ -94,6 +94,12 @@ impl NodeDirectory { { return Err(Error::Control("node session log recovery is incomplete")); } + crate::node::bundle::store::ensure_session_drained( + &self.layout, + current.session, + current.bundle, + ) + .await?; Ok(NodeSessionRecovery { fence: NodeSessionFence::from_record(¤t), log: current.log.clone(), @@ -122,6 +128,12 @@ impl NodeDirectory { { return Err(Error::Control("node session log retirement is incomplete")); } + crate::node::bundle::store::ensure_session_drained( + &self.layout, + current.session, + current.bundle, + ) + .await?; Ok(NodeSessionClosure { node: current.node, session: current.session, diff --git a/crates/cellule-runtime/src/node/directory/log.rs b/crates/cellule-runtime/src/node/directory/log.rs index 14363b75..ee67a373 100644 --- a/crates/cellule-runtime/src/node/directory/log.rs +++ b/crates/cellule-runtime/src/node/directory/log.rs @@ -197,6 +197,14 @@ impl NodeDirectory { now_ms: i64, ) -> Result { self.validate(&observed.advertisement, now_ms)?; + // This bounded catalog retains original binding tombstones for one + // boot/epoch. Clearing its head during rotation would erase Cell pins. + // Drain and withdraw this boot before starting another bundle lane. + if observed.advertisement.bundle.is_some() { + return Err(Error::Node( + "bundle-bound boot must drain before log rotation", + )); + } let current = observed .advertisement .log @@ -241,6 +249,12 @@ impl NodeDirectory { now_ms: i64, ) -> Result { self.validate(&observed.advertisement, now_ms)?; + crate::node::bundle::store::ensure_session_drained( + &self.layout, + observed.advertisement.session, + observed.advertisement.bundle, + ) + .await?; let current = observed .advertisement .log diff --git a/crates/cellule-runtime/src/node/directory/mod.rs b/crates/cellule-runtime/src/node/directory/mod.rs index ffe2e227..0f893538 100644 --- a/crates/cellule-runtime/src/node/directory/mod.rs +++ b/crates/cellule-runtime/src/node/directory/mod.rs @@ -254,8 +254,23 @@ impl NodeDirectory { return Ok(false); }; validate_record_path(&self.layout, record.session(), &path)?; - Ok(matches!(record, NodeRecord::Tombstone(current) - if current.claimant.is_none() && current.log.is_none())) + let NodeRecord::Tombstone(current) = record else { + return Ok(false); + }; + if current.claimant.is_some() || current.log.is_some() { + return Ok(false); + } + match crate::node::bundle::store::ensure_session_drained( + &self.layout, + current.session, + current.bundle, + ) + .await + { + Ok(()) => Ok(true), + Err(Error::PendingPublication) => Ok(false), + Err(source) => Err(source), + } } pub(super) fn validate(&self, advertisement: &NodeAdvertisement, now_ms: i64) -> Result<()> { @@ -496,6 +511,7 @@ pub(super) struct NodeTombstone { pub(super) claim_generation: u64, pub(super) claim_expires_at_ms: Option, pub(super) log: Option, + pub(super) bundle: Option, } impl NodeTombstone { @@ -506,6 +522,7 @@ impl NodeTombstone { retired_at_ms: i64, claimant: Option, log: Option, + bundle: Option, ) -> Result { let tombstone = Self { session, @@ -518,6 +535,7 @@ impl NodeTombstone { retired_at_ms.saturating_add(crate::node::log_state::RECOVERY_CLAIM_LIFETIME_MS) }), log, + bundle, }; tombstone.validate()?; Ok(tombstone) @@ -647,13 +665,17 @@ impl NodeTombstone { return Err(Error::Node("node tombstone exceeds 64 KiB")); } let raw: RawNodeTombstoneEnvelope = serde_json::from_slice(bytes)?; - if raw.tombstone.version != 1 { + if raw.tombstone.version != if raw.tombstone.bundle.is_some() { 2 } else { 1 } { return Err(Error::Node("unsupported node tombstone version")); } let raw = raw.tombstone; let session = SessionId::from_bytes(decode_hex(&raw.session)?); let node = NodeId::from_bytes(decode_hex(&raw.node)?); let tombstone = Self { + bundle: raw + .bundle + .map(crate::node::bundle::NodeBundleHead::try_from) + .transpose()?, session, node, expires_at_ms: canonical_i64(&raw.expires_at_ms)?, @@ -678,6 +700,9 @@ impl NodeTombstone { } pub(super) fn validate(&self) -> Result<()> { + if let Some(bundle) = self.bundle { + bundle.validate()?; + } if self.session.as_bytes().iter().all(|byte| *byte == 0) || self.node.as_bytes().iter().all(|byte| *byte == 0) || self.expires_at_ms < 0 diff --git a/crates/cellule-runtime/src/node/directory/recovery.rs b/crates/cellule-runtime/src/node/directory/recovery.rs index 458ff71b..a5304c2c 100644 --- a/crates/cellule-runtime/src/node/directory/recovery.rs +++ b/crates/cellule-runtime/src/node/directory/recovery.rs @@ -150,6 +150,7 @@ impl NodeDirectory { now_ms, None, advertisement.log.clone(), + advertisement.bundle, )? .claim(claimant, now_ms)? } diff --git a/crates/cellule-runtime/src/node/durability/mod.rs b/crates/cellule-runtime/src/node/durability/mod.rs index 3eaebe29..fa1b09af 100644 --- a/crates/cellule-runtime/src/node/durability/mod.rs +++ b/crates/cellule-runtime/src/node/durability/mod.rs @@ -208,15 +208,36 @@ impl NodeDurability { /// Assigns and asynchronously ships one captured commit to every member. pub async fn submit(&self, submission: NodeLogSubmission) -> Result { + self.submit_assigned(submission) + .await + .map(|(ticket, _)| ticket) + } + + /// Uses the same bounded native lane and retains the complete assigned range + /// witness required by shared bundle selection. + pub async fn submit_assigned( + &self, + submission: NodeLogSubmission, + ) -> Result<(CommitTicket, crate::node::log::AssignedCommitRange)> { self.node_lease.check()?; let ticket = tokio::select! { - result = self.shipper.submit(submission) => result?, + result = self.shipper.submit_assigned(submission) => result?, () = self.node_lease.wait_fenced() => return Err(Error::Fenced), }; self.node_lease.check()?; Ok(ticket) } + /// Freezes exact Cell issuance after the original SQL and capture tasks join. + pub fn close_cell_issuance( + &self, + scope: crate::node::log::CellLogScope, + base: cellule_ltx::RootRef, + ) -> Result { + self.node_lease.check()?; + self.gate.close_cell_issuance(scope, base) + } + /// Returns fleet proof only after follower fsync and authoritative activation. pub async fn prove_fleet(&self, ticket: CommitTicket) -> Result { self.node_lease.check()?; diff --git a/crates/cellule-runtime/src/node/log/mod.rs b/crates/cellule-runtime/src/node/log/mod.rs index 37dcd941..cf3c3e63 100644 --- a/crates/cellule-runtime/src/node/log/mod.rs +++ b/crates/cellule-runtime/src/node/log/mod.rs @@ -6,7 +6,7 @@ use std::sync::{Arc, Mutex}; use tokio::sync::Notify; use crate::identity::NodeId; -use crate::identity::SessionId; +use crate::identity::{ApplicationId, CellId, IncarnationId, SessionId}; use crate::node::log_transport::NodeLogTransport; use crate::node::{NodeDirectory, VersionedNodeAdvertisement}; use crate::{Error, Result}; @@ -22,6 +22,110 @@ pub use recovery::*; pub(crate) const MAX_TICKET_FRAMES: u64 = 1_024; +/// Exact Cell writer identity carried by the ordered native frame lane. +#[derive(Clone, Copy, Debug, PartialEq, Eq, Hash)] +pub struct CellLogScope { + /// Application that owns the Cell. + pub application: ApplicationId, + /// Cell whose issuance will be closed. + pub cell: CellId, + /// Original incarnation. + pub incarnation: IncarnationId, + /// Original writer epoch. + pub cell_epoch: u64, +} + +/// Frozen complete issued endpoint, including frames acknowledged by followers +/// above the selected object prefix. Only the ordered gate can construct it. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub struct CellIssuedRange { + leader_session: SessionId, + log_epoch: u64, + scope: CellLogScope, + last_node_sequence: u64, + commit_sequence: u64, + first_commit_sequence: u64, + position: cellule_ltx::Position, +} + +/// Exact complete capture assigned by the ordered native lane. A frame prefix +/// cannot substitute for this capability even when it carries the final logical +/// command number. Its digest binds every assigned native frame in order. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub struct AssignedCommitRange { + ticket: CommitTicket, + scope: CellLogScope, + first_commit: u64, + commit: u64, + position: cellule_ltx::Position, + digest: [u8; 32], +} +impl AssignedCommitRange { + /// Exact original native ticket. + pub const fn ticket(&self) -> CommitTicket { + self.ticket + } + pub(crate) fn verify(&self, frames: &[cellule_ltx::VerifiedNodeFrame]) -> Result<()> { + if frames.is_empty() + || frames.len() as u64 != self.ticket.last_sequence - self.ticket.first_sequence + 1 + { + return Err(Error::Node("bundle omits part of an assigned capture")); + } + let mut hash = blake3::Hasher::new(); + for (offset, frame) in frames.iter().enumerate() { + let scope = frame.scope(); + if scope.leader_session != *self.ticket.leader_session.as_bytes() + || scope.log_epoch != self.ticket.log_epoch + || scope.node_sequence != self.ticket.first_sequence + offset as u64 + || scope.application != *self.scope.application.as_bytes() + || scope.cell != *self.scope.cell.as_bytes() + || scope.incarnation != *self.scope.incarnation.as_bytes() + || scope.cell_epoch != self.scope.cell_epoch + || frame.first_commit_sequence() != self.first_commit + || scope.commit_sequence != self.commit + { + return Err(Error::Node("bundle assigned capture scope differs")); + } + hash.update(&frame.digest()); + } + if *hash.finalize().as_bytes() != self.digest + || frames + .last() + .is_none_or(|frame| frame.segment().position() != self.position) + { + return Err(Error::Node("bundle assigned capture digest differs")); + } + Ok(()) + } +} + +impl CellIssuedRange { + /// Original lane session. + pub const fn leader_session(&self) -> SessionId { + self.leader_session + } + /// Original lane epoch. + pub const fn log_epoch(&self) -> u64 { + self.log_epoch + } + /// Exact original writer. + pub const fn scope(&self) -> CellLogScope { + self.scope + } + /// Complete assigned native endpoint, rather than sampled follower coverage. + pub const fn last_node_sequence(&self) -> u64 { + self.last_node_sequence + } + /// Complete logical command endpoint. + pub const fn commit_sequence(&self) -> u64 { + self.commit_sequence + } + /// Exact SQLite position of the complete issued range. + pub const fn position(&self) -> cellule_ltx::Position { + self.position + } +} + /// One actor-issued consecutive frame range awaiting a durability proof. #[derive(Clone, Copy, Debug, PartialEq, Eq)] pub struct CommitTicket { @@ -178,6 +282,9 @@ struct GateState { fleet_active: bool, rotating: bool, fenced: bool, + cell_issued: HashMap, + closed_cells: HashSet, + untracked_issuance: bool, } impl DurabilityGate { @@ -235,6 +342,9 @@ impl DurabilityGate { fleet_active: false, rotating: false, fenced: false, + cell_issued: HashMap::new(), + closed_cells: HashSet::new(), + untracked_issuance: false, })), changed: Arc::new(Notify::new()), }) @@ -259,6 +369,10 @@ impl DurabilityGate { state.next_sequence = last_sequence .checked_add(1) .ok_or(Error::Node("node sequence overflow"))?; + // This legacy API has no Cell identity. Its ranges cannot establish a + // complete per-Cell closure endpoint; the canonical shipper uses the + // verified commit_frames path instead. + state.untracked_issuance = true; Ok(CommitTicket { leader_session: state.leader_session, log_epoch: state.log_epoch, @@ -290,7 +404,11 @@ impl DurabilityGate { }) } - pub(crate) fn commit(&self, ticket: CommitTicket) -> Result<()> { + pub(crate) fn commit_frames( + &self, + ticket: CommitTicket, + frames: &[cellule_ltx::VerifiedNodeFrame], + ) -> Result> { let mut state = self.lock()?; if state.fenced { return Err(Error::Fenced); @@ -307,11 +425,151 @@ impl DurabilityGate { { return Err(Error::Node("node-log ticket reservation changed")); } + // One submission is one complete Cell capture. Stage its endpoint on + // the stack, then install it only after every frame passed validation. + let mut issued: Option = None; + let mut assignment = None; + if !frames.is_empty() + && frames.len() as u64 != ticket.last_sequence - ticket.first_sequence + 1 + { + return Err(Error::Node("node-log assigned frame range differs")); + } + for (index, frame) in frames.iter().enumerate() { + let scope = frame.scope(); + let cell_scope = CellLogScope { + application: ApplicationId::from_bytes(scope.application), + cell: CellId::from_bytes(scope.cell), + incarnation: IncarnationId::from_bytes(scope.incarnation), + cell_epoch: scope.cell_epoch, + }; + match &mut assignment { + None => { + assignment = Some(AssignedCommitRange { + ticket, + scope: cell_scope, + first_commit: frame.first_commit_sequence(), + commit: scope.commit_sequence, + position: frame.segment().position(), + digest: [0; 32], + }) + } + Some(assignment) => { + if assignment.scope != cell_scope + || assignment.first_commit != frame.first_commit_sequence() + || assignment.commit != scope.commit_sequence + { + return Err(Error::Node( + "assigned capture changes Cell or command range", + )); + } + assignment.position = frame.segment().position(); + } + } + if state.closed_cells.contains(&cell_scope) { + return Err(Error::Fenced); + } + if scope.leader_session != *state.leader_session.as_bytes() + || scope.log_epoch != state.log_epoch + || scope.node_sequence != ticket.first_sequence + index as u64 + { + return Err(Error::Node("node-log assigned frame scope differs")); + } + if let Some(previous) = issued.as_ref() { + let continuing_group = scope.commit_sequence == previous.commit_sequence; + if (continuing_group + && frame.first_commit_sequence() != previous.first_commit_sequence) + || (!continuing_group + && frame.first_commit_sequence() + != previous + .commit_sequence + .checked_add(1) + .ok_or(Error::Node("Cell commit overflow"))?) + || frame.segment().min_txid + != previous + .position + .txid + .checked_add(1) + .ok_or(Error::Node("Cell TXID overflow"))? + || frame.segment().pre_checksum != previous.position.checksum + { + return Err(Error::Node("node-log Cell issuance has a gap")); + } + } else if state.cell_issued.get(&cell_scope).is_some_and(|previous| { + // Separate captures may have intervening object-only commands. + // Require forward issuance here; only the bundle verifier can + // establish exact continuity from its authority-pinned base. + scope.commit_sequence <= previous.commit_sequence + || frame.first_commit_sequence() <= previous.commit_sequence + || frame.segment().max_txid <= previous.position.txid + }) { + return Err(Error::Node("node-log Cell issuance did not advance")); + } + if !state.cell_issued.contains_key(&cell_scope) && state.cell_issued.len() >= 4_096 { + return Err(Error::Capacity("node-log Cell issuance bindings")); + } + issued = Some(CellIssuedRange { + leader_session: state.leader_session, + log_epoch: state.log_epoch, + scope: cell_scope, + last_node_sequence: scope.node_sequence, + commit_sequence: scope.commit_sequence, + first_commit_sequence: frame.first_commit_sequence(), + position: frame.segment().position(), + }); + } state.next_sequence = ticket .last_sequence .checked_add(1) .ok_or(Error::Node("node sequence overflow"))?; - Ok(()) + if let Some(issued) = issued { + state.cell_issued.insert(issued.scope, issued); + } + if let Some(assignment) = &mut assignment { + let mut hash = blake3::Hasher::new(); + for frame in frames { + hash.update(&frame.digest()); + } + assignment.digest = *hash.finalize().as_bytes(); + } + Ok(assignment) + } + + /// Closes this writer's sequence assignment in the same lock as native + /// ticket commit. Call after joining its accepted SQL/capture/submission + /// jobs. Late assignment fails without consuming a global sequence. + pub fn close_cell_issuance( + &self, + scope: CellLogScope, + base: cellule_ltx::RootRef, + ) -> Result { + let mut state = self.lock()?; + if state.fenced || state.untracked_issuance { + return Err(Error::Fenced); + } + if scope.cell_epoch == 0 + || scope.cell.as_bytes().iter().all(|byte| *byte == 0) + || base.cell != *scope.cell.as_bytes() + || base.incarnation != *scope.incarnation.as_bytes() + { + return Err(Error::Node("invalid Cell issuance closure")); + } + if !state.closed_cells.contains(&scope) && state.closed_cells.len() >= 4_096 { + return Err(Error::Capacity("node-log closed Cell bindings")); + } + state.closed_cells.insert(scope); + Ok(state + .cell_issued + .get(&scope) + .copied() + .unwrap_or(CellIssuedRange { + leader_session: state.leader_session, + log_epoch: state.log_epoch, + scope, + last_node_sequence: 0, + first_commit_sequence: base.commit_sequence, + commit_sequence: base.commit_sequence, + position: base.position, + })) } /// Returns this gate's immutable enrolled epoch, including after rotation. diff --git a/crates/cellule-runtime/src/node/log_shipper/mod.rs b/crates/cellule-runtime/src/node/log_shipper/mod.rs index 4ee9399b..520f3c3f 100644 --- a/crates/cellule-runtime/src/node/log_shipper/mod.rs +++ b/crates/cellule-runtime/src/node/log_shipper/mod.rs @@ -78,6 +78,10 @@ impl NodeLogSubmission { || first_commit_sequence > commit_sequence || commit_sequence > i64::MAX as u64 || cuts.segments.is_empty() + || cuts + .segments + .last() + .is_none_or(|segment| segment.info().position() != cuts.position) || encoded_bytes.is_none() { return Err(Error::Node("invalid node-log submission")); @@ -143,7 +147,7 @@ struct LoadedNodeLogSubmission { } impl LoadedNodeLogSubmission { - fn encode(self, ticket: CommitTicket) -> Result> { + fn encode(self, ticket: CommitTicket) -> Result> { self.frames .into_iter() .enumerate() @@ -154,10 +158,7 @@ impl LoadedNodeLogSubmission { .first_sequence() .checked_add(offset) .ok_or(Error::Node("node-log sequence overflow"))?; - frame - .with_node_sequence(node_sequence) - .map(|frame| frame.encoded().clone()) - .map_err(Error::from) + frame.with_node_sequence(node_sequence).map_err(Error::from) }) .collect() } @@ -277,6 +278,17 @@ impl NodeLogShipper { /// Queue, byte admission, disk reads, and canonical encoding happen before /// the ticket reservation commits, so failures cannot create a sequence gap. pub async fn submit(&self, submission: NodeLogSubmission) -> Result { + self.submit_assigned(submission) + .await + .map(|(ticket, _)| ticket) + } + + /// Assigns the same canonical submission and returns its complete native + /// range witness for node-wide object publication. There is one shipping lane. + pub async fn submit_assigned( + &self, + submission: NodeLogSubmission, + ) -> Result<(CommitTicket, crate::node::log::AssignedCommitRange)> { let frame_count = submission.frame_count()?; if submission .segments @@ -317,21 +329,24 @@ impl NodeLogShipper { let _ordered = self.order.lock().await; let ticket = self.gate.preview(frame_count)?; let encoded = loaded.encode(ticket)?; - self.gate.commit(ticket)?; + let assignment = self + .gate + .commit_frames(ticket, &encoded)? + .ok_or(Error::Node("assigned capture is empty"))?; let reservation = Arc::new(OutstandingBytes { _permit: reservation, }); let frames = encoded .into_iter() .enumerate() - .map(|(offset, encoded)| QueuedFrame { + .map(|(offset, frame)| QueuedFrame { sequence: ticket.first_sequence().saturating_add(offset as u64), - encoded, + encoded: frame.encoded().clone(), _reservation: Arc::clone(&reservation), }) .collect(); slot.send(QueuedSubmission { frames }); - Ok(ticket) + Ok((ticket, assignment)) } /// Closes admission and drains every accepted frame to the current epoch. diff --git a/crates/cellule-runtime/src/node/log_shipper/tests.rs b/crates/cellule-runtime/src/node/log_shipper/tests.rs index b2c1173a..78e3bf82 100644 --- a/crates/cellule-runtime/src/node/log_shipper/tests.rs +++ b/crates/cellule-runtime/src/node/log_shipper/tests.rs @@ -190,6 +190,51 @@ fn submission(cuts: &cellule_ltx::CaptureBatch) -> NodeLogSubmission { .unwrap() } +#[tokio::test] +async fn intervening_object_only_commands_do_not_disable_native_issuance() { + let directory = tempfile::TempDir::new().unwrap(); + let mut db = cellule_ltx::Db::open( + &directory.path().join("object-gap.sqlite"), + cellule_ltx::Limits::default(), + ) + .unwrap(); + db.transaction(|tx| tx.execute_batch("CREATE TABLE items(value)")) + .unwrap(); + let first = db.capture().unwrap(); + db.transaction(|tx| tx.execute_batch("INSERT INTO items VALUES (5)")) + .unwrap(); + let _object_only = db.capture().unwrap(); + db.transaction(|tx| tx.execute_batch("INSERT INTO items VALUES (6)")) + .unwrap(); + let last = db.capture().unwrap(); + let gate = DurabilityGate::new(session(1), node(1), 2, [node(2)]).unwrap(); + gate.activate_fleet().unwrap(); + let shipper = NodeLogShipper::new_with_telemetry( + gate.clone(), + Arc::new(RecordingTransport::default()), + cellule_ltx::Limits::default(), + crate::fleet::telemetry::CellTelemetryHandle::default(), + ) + .unwrap(); + let first = shipper.submit(submission(&first)).await.unwrap(); + let next = NodeLogSubmission::new( + ApplicationId::from_bytes([9; 16]), + CellId::from_bytes([8; 32]), + IncarnationId::from_bytes([7; 16]), + 3, + 6, + &last, + ) + .unwrap(); + let (ticket, _) = shipper.submit_assigned(next).await.unwrap(); + assert_eq!(ticket.first_sequence(), first.last_sequence() + 1); + assert_eq!( + gate.prove(ticket).await.unwrap().source(), + crate::node::log::DurabilitySource::Fleet + ); + shipper.shutdown().await.unwrap(); +} + #[tokio::test] async fn concurrent_submissions_stay_ordered_and_require_every_member_ack() { let (_directory, cuts) = capture(); @@ -207,7 +252,17 @@ async fn concurrent_submissions_stay_ordered_and_require_every_member_ack() { let (first, second) = tokio::join!( shipper.submit(submission(&cuts)), - shipper.submit(submission(&cuts)) + shipper.submit( + NodeLogSubmission::new( + ApplicationId::from_bytes([9; 16]), + CellId::from_bytes([6; 32]), + IncarnationId::from_bytes([7; 16]), + 3, + 4, + &cuts + ) + .unwrap() + ) ); let first = first.unwrap(); let second = second.unwrap(); @@ -249,8 +304,25 @@ async fn append_telemetry_records_one_result_for_each_batch() { #[tokio::test] async fn splits_large_submission_at_sixty_four_frames() { - let (_directory, mut cuts) = capture(); - cuts.segments = std::iter::repeat_n(cuts.segments[0].clone(), 65).collect(); + let directory = tempfile::TempDir::new().unwrap(); + let mut database = cellule_ltx::Db::open( + &directory.path().join("many-cuts.sqlite"), + cellule_ltx::Limits::default(), + ) + .unwrap(); + database + .transaction(|tx| tx.execute_batch("CREATE TABLE items(value)")) + .unwrap(); + let mut cuts = database.capture().unwrap(); + for value in 0..64 { + database + .transaction(|tx| tx.execute("INSERT INTO items VALUES (?1)", [value])) + .unwrap(); + let next = database.capture().unwrap(); + cuts.segments.extend(next.segments); + cuts.position = next.position; + } + assert_eq!(cuts.segments.len(), 65); let gate = DurabilityGate::new(session(1), node(1), 2, [node(2)]).unwrap(); gate.activate_fleet().unwrap(); let transport = Arc::new(RecordingTransport::default()); @@ -310,7 +382,7 @@ async fn covered_queued_prefix_keeps_the_uncovered_suffix_fleet_durable() { .enumerate() .map(|(offset, encoded)| QueuedFrame { sequence: offset as u64 + 1, - encoded, + encoded: encoded.encoded().clone(), _reservation: Arc::clone(&reservation), }) .collect(); diff --git a/crates/cellule-runtime/src/node/mod.rs b/crates/cellule-runtime/src/node/mod.rs index f72d14e7..1af176ed 100644 --- a/crates/cellule-runtime/src/node/mod.rs +++ b/crates/cellule-runtime/src/node/mod.rs @@ -1,5 +1,6 @@ //! Node advertisements, capacity, the node directory, leases, and the durability log. pub mod append_grant; +pub mod bundle; pub mod durability; pub mod lease; pub mod log; diff --git a/crates/cellule-runtime/src/publication/mod.rs b/crates/cellule-runtime/src/publication/mod.rs index ad80c213..d6c56900 100644 --- a/crates/cellule-runtime/src/publication/mod.rs +++ b/crates/cellule-runtime/src/publication/mod.rs @@ -100,6 +100,37 @@ impl CellPublisher { self } + /// Materializes a selected exact bundle prefix through the ordinary root, + /// lineage and fenced Cell CAS path. The proof retains bounded locators; + /// materialization owns fresh I/O rather than retained capture bodies. + pub async fn materialize_bundle( + &mut self, + proof: &crate::node::bundle::BundleCoverageProof, + ) -> Result { + self.check_node_lease()?; + if self.observed.value().bundle_binding != Some(proof.binding()) { + return Err(Error::Fenced); + } + let base = self.observed.value().ltx_root().ok_or(Error::Fenced)?; + if base.commit_sequence == proof.commit_sequence() && base.position == proof.position() { + return Ok(base); + } + let overlay = proof + .recovery_overlay_from(self.authority.layout(), self.replica.limits(), base) + .await?; + self.check_node_lease()?; + let (replica, confirmation) = lineage::replica(self.replica.clone(), &self.authority); + let prepared = replica + .prepare_recovered_overlay(&overlay, self.observed.value().schema) + .await + .map_err(lineage::error)?; + self.lineage_confirmed = *confirmation + .lock() + .map_err(|_| Error::Peer("root lineage confirmation lock poisoned"))?; + self.publish_prepared(&prepared, self.observed.value().next_due_ms) + .await + } + pub(crate) fn with_shared_publication( mut self, coordinator: std::sync::Arc, diff --git a/crates/cellule-runtime/src/recovery/backup/mod.rs b/crates/cellule-runtime/src/recovery/backup/mod.rs index 9741c379..663c66eb 100644 --- a/crates/cellule-runtime/src/recovery/backup/mod.rs +++ b/crates/cellule-runtime/src/recovery/backup/mod.rs @@ -366,6 +366,9 @@ impl BackupPinStore { } async fn verify_root(&self, control: &Control) -> Result<()> { + if control.bundle_binding.is_some() { + return Err(Error::Backup("bundle-bound Cell must drain before backup")); + } if let Some(root) = control.ltx_root() { cellule_ltx::CellReplica::new( self.layout.clone(), diff --git a/crates/cellule-runtime/src/recovery/retention/tests.rs b/crates/cellule-runtime/src/recovery/retention/tests.rs index b4c5e728..755775ff 100644 --- a/crates/cellule-runtime/src/recovery/retention/tests.rs +++ b/crates/cellule-runtime/src/recovery/retention/tests.rs @@ -100,6 +100,7 @@ fn idle_control( owner: None, root: Some(RootRef::from_ltx(cell, incarnation, root).unwrap()), recovery: None, + bundle_binding: None, code, schema: 1, next_due_ms: None, diff --git a/docs/bundle-coverage-implementation.md b/docs/bundle-coverage-implementation.md new file mode 100644 index 00000000..78267ba4 --- /dev/null +++ b/docs/bundle-coverage-implementation.md @@ -0,0 +1,140 @@ +# Bundle coverage implementation + +The connected protocol APIs now implement shared selection, independently +awaitable root materialization and complete live-writer closure. They are **not +enabled in the ordinary actor response path**. The prior performance regression +and failed qualification remain the baseline. This slice establishes ordering +and reconstruction evidence; it makes no new throughput or latency claim. + +## Celld reference and Cellule adaptation + +The reference is celld `f2bf648663a610eefde71f3547ad61e9b896b1f0`. +Its [bundle loop](https://github.com/denoland/celld/blob/f2bf648663a610eefde71f3547ad61e9b896b1f0/crates/celld/ltx_repl.rs#L4401) +gathers Cell tails into one node upload and separates durable coverage from +materialized positions. Its [bundle credit path](https://github.com/denoland/celld/blob/f2bf648663a610eefde71f3547ad61e9b896b1f0/crates/celld/node_log.rs#L6620) +rechecks bucket state and epoch after the upload before crediting rows. +These are architectural references; this module is independently implemented. + +Cellule also has independent Cell ownership epochs. It pins each original +writer in Cell control before enrolling it in the complete node catalog. The +existing canonical node-record CAS selects catalog and range together. A +heartbeat preserves the head; a conflicting closure requires a fresh proposal. +Uploading an immutable object supplies no coverage proof. + +## Implemented contracts + +| API or boundary | Behavior | +| --- | --- | +| `NodeDirectory::initialize_bundle_lane` / `bind_bundle_cell` | Establish a boot/epoch catalog and pin an original serving Cell's base, code/schema and writer identity before issuing bundled commands | +| `NodeLogShipper::submit_assigned` / `NodeDurability::submit_assigned` | Use the existing bounded native shipping lane and return an opaque witness for every frame in a complete capture | +| `prepare_node_bundle` / `select_node_bundle` | Reject missing, overlapping, unassigned or cross-binding ranges; verify origin extents before selecting; reconcile only the exact head under the original lease | +| `BundleCoverageProof` | Retain authenticated immutable locators, logical endpoint, SQLite position and original base; retain no capture bodies | +| `load_bundle_coverage` | Reopen a selected suffix from the authority-pinned canonical catalog, including a fenced boot; grant no writer or follower-suffix closure | +| `CellPublisher::materialize_bundle` | Reconstruct the exact overlay and use normal root preparation, lineage and Cell CAS independently of selection | +| `checkpoint_bundle_cell` | Read current Cell authority and drop only an exact complete materialized prefix; retain newer selected suffix locators | +| `close_cell_issuance` / `begin_bundle_close` / `finish_bundle_close` | Freeze complete assigned issuance, including prior Fleet ACKs above the selected prefix; permit only its old issued tail to drain; close only at the exact terminal endpoint | +| Cell departure CAS | Refuse release, takeover and tombstone until Closed, exact materialization and catalog checkpoint; migration refuses while bound | +| Node withdrawal/maintenance | Refuse unresolved bindings; stale fencing preserves the catalog head; one bundle-bound boot cannot rotate its native log to another epoch | +| Backup and collection | Backup refuses bound Cells. Coverage objects have no deletion path; this is retention, not a qualified collection implementation | + +The original SQL/capture/submission jobs must join before `close_cell_issuance`. +Its ordered gate prevents late assignment from consuming a node sequence. The +legacy identity-free `DurabilityGate::issue` cannot produce a per-Cell closure: +using it makes that closure fail closed. Automatic actor joining and scheduling +are still required before enabling this path for application commands. + +```mermaid +sequenceDiagram + participant Cell as Original Cell writer + participant Gate as Ordered native gate + participant Node as Canonical node record + participant Root as Cell root materializer + Cell->>Gate: Complete capture assignment + Cell->>Node: One verified cross-Cell bundle + CAS + Node-->>Cell: Exact selected range proof + Node->>Root: Base + bounded authenticated locators + Root->>Cell: Ordinary exact root/lineage CAS + Cell->>Gate: Join accepted jobs, close assignment + Gate-->>Node: Complete issued terminal endpoint + Node->>Node: Closing, drain exact issued tail, Closed + Node->>Root: Materialize complete frozen terminal + Root->>Cell: Final ordinary root/lineage CAS + Root->>Node: Exact root checkpoint + Cell->>Cell: Release/transfer CAS may proceed +``` + +## Format and bounds + +Unbound Cell controls and node records retain version-1 encodings. A binding or +bundle head selects version 2; inconsistent version/field combinations and +unknown fields are rejected. Old readers must not operate on those records. +Use a fresh development prefix and initialize the lane before native issuance; +there is no live-format migration or same-boot bundle-epoch rotation. + +The `CNB1` object embeds a complete catalog and native frame extents under +`cells/v1/node-logs///coverage/v1/.cnb`. It is outside the +advertisement scan prefix. Both manifests and individual frame extents are +authenticated. New-object locators use a canonical self reference, resolved +against the selected digest, avoiding a self-referential digest field. + +| Development bound | Value | +| --- | ---: | +| Immutable object/catalog bytes | 4 MiB | +| Original bindings per boot/epoch | 4,096 | +| Native frames per selection | 64 | +| Uncheckpointed locators per Cell | 32 | +| Encoded suffix bytes verified per Cell | 4 MiB | +| Distinct base-root origin dependencies verified per Cell | 65,536 | + +Exhaustion rejects further bundle preparation while retaining the last proof. +These are safety ceilings, **not performance-qualified policies**. The complete +catalog is rewritten per selection and old locator bodies are reverified. +This is not yet the file-backed index needed for the 160/214-command checkpoint +cost constraints in the [authority decision](bundle-coverage-proof.md#quantified-checkpoint-constraint). +The caller owns host memory/native-job admission; automatic materializer budget, +fairness and cancellation/drain ownership are not integrated. + +## Verification and remaining delivery + +Twenty focused tests cover real managed SQLite captures, canonical object-store +CAS and materialization: two-Cell shared selection, byte-identical roots, a prior +Fleet ACK above the selected prefix, held old proposals across closure, hot +siblings and dormant bindings, lost replies, lease loss, partial command groups, +overlaps, corrupt/missing objects, version rejection, checkpoint continuation, +locator pressure and materializer failure. One test removes the original +database/captures, fences the node record and reconstructs the selected outcome +from origin. Its Fleet ACK setup uses the native gate; it is not a physical +follower-fsync qualification. + +The asynchronous tests select another command while an older root is pending. +The older root checkpoints a complete prefix without dropping the hot suffix; +the next materializer can continue from that root before or after the catalog +checkpoint. Origin lookup also works between the root CAS and catalog CAS. + +The two-Cell selection test observes **one immutable PUT plus one node CAS**, +with both Cell roots unchanged. That excludes enrollment, root materialization, +catalog checkpoints, compaction and maintenance and must not be reported as +total PUTs/command, TPS or a latency result. + +The isolated source snapshot passed all contributor checks: **1,884 workspace +tests passed, 38 ignored; 58 local LTX tests passed**. All-feature/all-target +checking, warnings-denied Clippy and API docs, formatting, boundaries, module +ownership, document syntax/links and SQL/peer validation also passed. Ignored +environment-dependent tests and the complete production qualification remain +outside this result. + +Remaining work before responses can use bundle proof: + +1. Integrate actor/executor proof and query/retry endpoints, release proved + capture retention, and schedule admitted materializers with joined shutdown. +2. Replace complete catalog rewrites with a bounded authenticated file-backed + lookup/checkpoint index; measure the full checkpoint and collection cost. +3. Fence and seal a failed original node's complete follower-issued suffix, + including ACKs above the selected head, before reconstructing and closing + every original binding. Current departure/maintenance guards refuse this + unfinished recovery rather than advancing a replacement writer. +4. Implement complete cross-Cell reference inventory, pins and grace-qualified + collection. Preserve original catalog and base dependencies throughout. +5. Run the unchanged all-ACK cold recovery, paired Docker throughput/latency, + debt, read, overload and drain qualification from the + [performance proposal](write-performance-proposal.md). diff --git a/docs/bundle-coverage-proof.md b/docs/bundle-coverage-proof.md index 43a03b41..2c31467d 100644 --- a/docs/bundle-coverage-proof.md +++ b/docs/bundle-coverage-proof.md @@ -1,9 +1,10 @@ # Bundle coverage authority decision -Status: protocol model and implementation design; bundle-based bucket responses -are disabled. Shared payload upload and signed append grants are separate -implemented paths. The [proposal](write-performance-proposal.md) remains the -qualification contract. +Status: protocol model and connected implementation APIs; bundle-based bucket +responses remain disabled. [Implementation and remaining gates](bundle-coverage-implementation.md) +records the verified slice and its limits. Shared payload upload and signed +append grants are separate implemented paths. The +[proposal](write-performance-proposal.md) remains the qualification contract. ## Cost and latency gap diff --git a/docs/write-performance-delivery.md b/docs/write-performance-delivery.md index f37afff3..117990f2 100644 --- a/docs/write-performance-delivery.md +++ b/docs/write-performance-delivery.md @@ -40,7 +40,7 @@ collection paths. There is no legacy decoding or automatic migration. | M1 | Packs, inline leaves and bounded compaction spooling delivered | Ordinary append meets four PUTs; composed compaction needs five. Paired cold/sparse-read guardrail unverified | | M2 | Bounded file-backed shared publication coordinator implemented; exact scope, restore, cancellation, minimum-budget and dormant-sibling retention checks added | Latest three active-Fleet windows cost 5.229–5.433 PUTs/command; two fail the debt trend. Three per-Cell authority PUTs remain; M4 is required | | M3 | Signed 512-sequence/five-second grants, bounded local registry, lifecycle gate and signed HTTP fixture implemented | Full isolated checks and native lifecycle suite pass; latest three active-Fleet windows cost 0.0138 enrollment GETs/command. The 15K target diagnostic fails delivery and warm audit | -| M4 | Binding/selector and delayed-materialization models plus [authority decision](bundle-coverage-proof.md) delivered | Production bundle proof, atomic transfer/recovery/collection and bundle ACKs not implemented | +| M4 | Models and [connected protocol APIs](bundle-coverage-implementation.md) for shared selection, exact assigned ranges, independent materialization, checkpoint and complete live-writer closure | Actor response/read integration, failed-node issued-suffix recovery, bounded index/host admission and bundle collection remain incomplete; bundle ACKs disabled | | M5 | Three paired low-rate Fleet repetitions and target diagnostics with exact ACK audits delivered | Publication stability and target delivery fail; qualified capacity, read/failure/overload matrix and absolute/relative parity remain unverified | ## Shared publication checkpoint @@ -267,7 +267,7 @@ repetitions at 100/s are diagnostic points rather than a capacity search. | Three stable Fleet windows and M2's 0.25 PUT/command budget | Fail: two candidate debt trends grow; cost 5.229–5.433 | | 15K Fleet and 2K bucket absolute targets | Fail in every arm; candidate also completes fewer overloaded commands than main | | M3's 0.05 fresh enrollment GET budget | Within budget at the three 100/s points; full qualification remains unverified | -| M4's production proof, atomic fault matrix and 0.05 total PUT budget | Unimplemented; abstract models do not enable ACKs or GC | +| M4's complete production integration, atomic fault matrix and 0.05 total PUT budget | Incomplete; protocol tests do not enable ACKs or GC and no new TPS/cost qualification has passed | | Read-only/mixed capacity and 1% hot-Cell guardrails | Unverified in Docker; two native routing p99 ratios exceed 1.2 | | A/A capacity variance and relative parity | Unverified; overloaded completions cannot supply the reference | | Qualified overload, safe refusal before SQL and immediate recovery | Unverified; the historical step-down failed latency | From 7fc079345b5214a18a001833732a84039e4485db Mon Sep 17 00:00:00 2001 From: forhappy Date: Wed, 7 Oct 2026 09:39:16 -0700 Subject: [PATCH 015/102] Reserve bundle inventory before pinning Cell authority --- .../src/control/authority/mod.rs | 3 + .../src/node/bundle/binding.rs | 147 +++++++++------ .../src/node/bundle/closure.rs | 1 + .../cellule-runtime/src/node/bundle/codec.rs | 2 + crates/cellule-runtime/src/node/bundle/mod.rs | 3 + .../cellule-runtime/src/node/bundle/proof.rs | 3 + .../src/node/bundle/selection.rs | 2 +- .../cellule-runtime/src/node/bundle/store.rs | 88 +++++++-- .../src/node/bundle/tests/faults.rs | 170 ++++++++++++++++++ .../src/node/bundle/tests/mod.rs | 19 +- docs/bundle-coverage-implementation.md | 19 +- docs/bundle-coverage-proof.md | 2 +- 12 files changed, 373 insertions(+), 86 deletions(-) diff --git a/crates/cellule-runtime/src/control/authority/mod.rs b/crates/cellule-runtime/src/control/authority/mod.rs index 738bfdce..1befb383 100644 --- a/crates/cellule-runtime/src/control/authority/mod.rs +++ b/crates/cellule-runtime/src/control/authority/mod.rs @@ -177,6 +177,9 @@ impl CellAuthority { transition: Transition, ) -> Result { observed.value.validate_transition(&next, transition)?; + if transition == Transition::BindBundle { + crate::node::bundle::store::ensure_enrollment(&self.layout, &next).await?; + } if observed.value.bundle_binding.is_some() && observed.value.bundle_binding != next.bundle_binding { diff --git a/crates/cellule-runtime/src/node/bundle/binding.rs b/crates/cellule-runtime/src/node/bundle/binding.rs index 29461aac..a74fea2b 100644 --- a/crates/cellule-runtime/src/node/bundle/binding.rs +++ b/crates/cellule-runtime/src/node/bundle/binding.rs @@ -35,8 +35,9 @@ impl NodeDirectory { self.select_catalog(observed, &prepared, now_ms).await } - /// Pins one original writer in Cell authority before enrolling it in the - /// complete catalog. An ambiguous failure retains the pin and blocks departure. + /// Reserves a provisional catalog entry before pinning the original Cell, + /// then opens it only after the Cell CAS is confirmed. Cancellation or an + /// ambiguous failure leaves a catalog obligation that blocks maintenance. pub async fn bind_bundle_cell( &self, observed: &VersionedNodeAdvertisement, @@ -68,17 +69,60 @@ impl NodeDirectory { return Err(Error::Fenced); } let application = ApplicationId::from_bytes(*authority.layout().application_id()); - let pinned = if let Some(pin) = value.bundle_binding { - if pin.session != catalog.session || pin.epoch != catalog.epoch { + let existing = catalog.bindings.iter().find(|binding| { + binding.application == application + && binding.control.cell == value.cell + && binding.phase != BindingPhase::Closed + }); + if let Some(binding) = existing { + if binding.control.incarnation != value.incarnation + || binding.control.epoch != value.epoch + || binding.control.owner != value.owner + || binding.control.code != value.code + || binding.control.schema != value.schema + || binding.control.root != value.root + { return Err(Error::Fenced); } - control.clone() + if binding.phase == BindingPhase::Open { + if binding.control.bundle_binding != value.bundle_binding { + return Err(Error::Fenced); + } + return Ok((observed.clone(), control.clone())); + } + if binding.phase != BindingPhase::Provisional { + return Err(Error::Fenced); + } + } + let pin = if let Some(binding) = existing { + binding + .control + .bundle_binding + .ok_or(Error::Node("provisional binding lacks pin"))? + } else if let Some(pin) = value.bundle_binding { + pin } else { let mut identity = value.encode()?; identity.extend_from_slice(b"cellule.bundle-binding.v1\0"); identity.extend_from_slice(application.as_bytes()); identity.extend_from_slice(&catalog.epoch.to_le_bytes()); - let mut next = value.clone(); + BundleBindingRef { + session: catalog.session, + epoch: catalog.epoch, + digest: Digest::from_bytes(*blake3::hash(&identity).as_bytes()), + } + }; + if pin.session != catalog.session + || pin.epoch != catalog.epoch + || value.bundle_binding.is_some_and(|current| current != pin) + || catalog.bindings.iter().any(|binding| { + binding.phase == BindingPhase::Closed && binding.control.bundle_binding == Some(pin) + }) + { + return Err(Error::Fenced); + } + let mut next = value.clone(); + if next.bundle_binding.is_none() { next.revision = next .revision .checked_add(1) @@ -87,11 +131,41 @@ impl NodeDirectory { .progress .checked_add(1) .ok_or(Error::Control("progress overflow"))?; - next.bundle_binding = Some(BundleBindingRef { - session: catalog.session, - epoch: catalog.epoch, - digest: Digest::from_bytes(*blake3::hash(&identity).as_bytes()), + next.bundle_binding = Some(pin); + } + // Select the inventory obligation first. Otherwise cancellation after + // Cell pinning could leave a pin absent from the complete node catalog. + let reserved = if existing.is_none() { + let base = next + .ltx_root() + .ok_or(Error::Node("bundle binding lacks published base"))?; + catalog.bindings.push(Binding { + application, + first_commit: base.commit_sequence, + control: next.clone(), + phase: BindingPhase::Provisional, + terminal: None, + selected_sequence: 0, + selected_commit: base.commit_sequence, + selected_position: base.position, + locators: Vec::new(), }); + catalog.bindings.sort_unstable_by_key(|binding| { + binding + .control + .bundle_binding + .map(|pin| *pin.digest.as_bytes()) + }); + let prepared = self + .upload_catalog(Some(head), catalog.clone(), &[]) + .await?; + self.select_catalog(observed, &prepared, now_ms).await? + } else { + observed.clone() + }; + let pinned = if value.bundle_binding.is_some() { + control.clone() + } else { match authority .transition(control, next.clone(), Transition::BindBundle) .await @@ -103,53 +177,14 @@ impl NodeDirectory { }, } }; - let pin = pinned - .value() - .bundle_binding - .ok_or(Error::Node("Cell bundle pin missing"))?; - if let Some(binding) = catalog - .bindings - .iter() - .find(|binding| binding.control.bundle_binding == Some(pin)) - { - if binding.phase != BindingPhase::Open || binding.control != *pinned.value() { - return Err(Error::Fenced); - } - return Ok((observed.clone(), pinned)); - } - // Enrolling two writer bindings for one Cell would allow a departed - // epoch to rejoin under a new identity. Closed originals remain tombstones. - if catalog.bindings.iter().any(|binding| { - binding.application == application - && binding.control.cell == value.cell - && binding.phase != BindingPhase::Closed - }) { - return Err(Error::Fenced); - } - let base = pinned - .value() - .ltx_root() - .ok_or(Error::Node("bundle binding lacks published base"))?; - catalog.bindings.push(Binding { - application, - first_commit: base.commit_sequence, - control: pinned.value().clone(), - phase: BindingPhase::Open, - terminal: None, - selected_sequence: 0, - selected_commit: base.commit_sequence, - selected_position: base.position, - locators: Vec::new(), - }); - catalog.bindings.sort_unstable_by_key(|binding| { - binding - .control - .bundle_binding - .map(|pin| *pin.digest.as_bytes()) - }); - let prepared = self.upload_catalog(Some(head), catalog, &[]).await?; + let binding = catalog.binding_mut(pin.digest)?; + binding.control = pinned.value().clone(); + binding.phase = BindingPhase::Open; + let prepared = self + .upload_catalog(reserved.advertisement.bundle, catalog, &[]) + .await?; Ok(( - self.select_catalog(observed, &prepared, now_ms).await?, + self.select_catalog(&reserved, &prepared, now_ms).await?, pinned, )) } diff --git a/crates/cellule-runtime/src/node/bundle/closure.rs b/crates/cellule-runtime/src/node/bundle/closure.rs index c20b84db..38b0e782 100644 --- a/crates/cellule-runtime/src/node/bundle/closure.rs +++ b/crates/cellule-runtime/src/node/bundle/closure.rs @@ -93,6 +93,7 @@ impl NodeDirectory { || scope.incarnation != binding.control.incarnation || scope.cell_epoch != binding.control.epoch || binding.phase == BindingPhase::Closed + || binding.phase == BindingPhase::Provisional { return Err(Error::Fenced); } diff --git a/crates/cellule-runtime/src/node/bundle/codec.rs b/crates/cellule-runtime/src/node/bundle/codec.rs index 34f3a3e7..01ba2770 100644 --- a/crates/cellule-runtime/src/node/bundle/codec.rs +++ b/crates/cellule-runtime/src/node/bundle/codec.rs @@ -33,6 +33,7 @@ fn metadata(catalog: &Catalog, frames: usize) -> Result { e.write_u64(binding.first_commit)?; e.write_bytes(&binding.control.encode()?)?; e.write_u8(match binding.phase { + BindingPhase::Provisional => 3, BindingPhase::Open => 0, BindingPhase::Closing => 1, BindingPhase::Closed => 2, @@ -127,6 +128,7 @@ pub(super) fn decode(body: &Bytes) -> Result { let first_commit = d.read_u64()?; let control = Control::decode(d.read_bytes()?)?; let phase = match d.read_u8()? { + 3 => BindingPhase::Provisional, 0 => BindingPhase::Open, 1 => BindingPhase::Closing, 2 => BindingPhase::Closed, diff --git a/crates/cellule-runtime/src/node/bundle/mod.rs b/crates/cellule-runtime/src/node/bundle/mod.rs index 92a20e6f..c03638a2 100644 --- a/crates/cellule-runtime/src/node/bundle/mod.rs +++ b/crates/cellule-runtime/src/node/bundle/mod.rs @@ -85,6 +85,7 @@ impl NodeBundleHead { #[derive(Clone, Copy, Debug, PartialEq, Eq)] enum BindingPhase { + Provisional, Open, Closing, Closed, @@ -221,6 +222,8 @@ impl Catalog { return Err(Error::Node("invalid bundle binding coverage")); } match (binding.phase, binding.terminal) { + (BindingPhase::Provisional, None) + if binding.locators.is_empty() && binding.selected_sequence == 0 => {} (BindingPhase::Open, None) => {} (BindingPhase::Closing, Some((sequence, commit, position))) if sequence >= binding.selected_sequence diff --git a/crates/cellule-runtime/src/node/bundle/proof.rs b/crates/cellule-runtime/src/node/bundle/proof.rs index 515232bc..58b2909b 100644 --- a/crates/cellule-runtime/src/node/bundle/proof.rs +++ b/crates/cellule-runtime/src/node/bundle/proof.rs @@ -43,6 +43,9 @@ impl NodeDirectory { let head = head.ok_or(Error::PendingPublication)?; let mut catalog = load_catalog(&self.layout, session, head).await?; let binding = catalog.binding_mut(pin.digest)?.clone(); + if binding.phase == BindingPhase::Provisional { + return Err(Error::PendingPublication); + } if head.epoch != pin.epoch || binding.control.bundle_binding != Some(pin) || binding.application.as_bytes() != authority.layout().application_id() diff --git a/crates/cellule-runtime/src/node/bundle/selection.rs b/crates/cellule-runtime/src/node/bundle/selection.rs index 8c655faf..194161a1 100644 --- a/crates/cellule-runtime/src/node/bundle/selection.rs +++ b/crates/cellule-runtime/src/node/bundle/selection.rs @@ -61,7 +61,7 @@ impl NodeDirectory { && binding.control.epoch == scope.cell_epoch }) .ok_or(Error::Node("bundle row has no enrolled Cell binding"))?; - if binding.phase == BindingPhase::Closed + if !matches!(binding.phase, BindingPhase::Open | BindingPhase::Closing) || binding.locators.len() >= MAX_LOCATORS || binding .terminal diff --git a/crates/cellule-runtime/src/node/bundle/store.rs b/crates/cellule-runtime/src/node/bundle/store.rs index 94082534..a6903882 100644 --- a/crates/cellule-runtime/src/node/bundle/store.rs +++ b/crates/cellule-runtime/src/node/bundle/store.rs @@ -1,7 +1,7 @@ //! Canonical immutable catalog I/O and departure guards. use super::*; use crate::node::directory::NodeRecord; -use crate::node::{NodeDirectory, VersionedNodeAdvertisement}; +use crate::node::{NodeAdvertisement, NodeDirectory, VersionedNodeAdvertisement}; impl NodeDirectory { pub(super) async fn upload_catalog( @@ -55,16 +55,9 @@ impl NodeDirectory { if *blake3::hash(&prepared.body).as_bytes() != *prepared.head.digest.as_bytes() { return Err(Error::Node("bundle proposal digest differs")); } + let mut last_conflict = None; for _ in 0..4 { - self.validate(&base.advertisement, now_ms)?; - if base.advertisement.session != prepared.catalog.session - || base.advertisement.log.as_ref().is_some_and(|log| { - log.epoch() != prepared.head.epoch - || log.phase() != crate::node::log_state::NodeLogPhase::Open - }) - { - return Err(Error::Fenced); - } + self.validate_bundle_source(&base.advertisement, prepared, now_ms)?; if base.advertisement.bundle == Some(prepared.head) { return Ok(base); } @@ -80,17 +73,40 @@ impl NodeDirectory { match self.update_advertisement(&base, next, now_ms).await { Ok(selected) => return Ok(selected), Err(source) => match self.load(prepared.catalog.session, now_ms).await { + Ok(Some(current)) if current.advertisement.bundle == Some(prepared.head) => { + self.validate_bundle_source(¤t.advertisement, prepared, now_ms)?; + return Ok(current); + } Ok(Some(current)) - if current.advertisement.bundle == Some(prepared.head) - || current.advertisement.bundle == prepared.original => + if current.advertisement.bundle == prepared.original + && current.advertisement != base.advertisement => { - base = current + last_conflict = Some(source); + base = current; } _ => return Err(source), }, } } - Err(Error::Node("bundle selection CAS contention")) + Err(last_conflict.unwrap_or(Error::Node("bundle selection CAS contention"))) + } + + fn validate_bundle_source( + &self, + source: &NodeAdvertisement, + prepared: &PreparedNodeBundle, + now_ms: i64, + ) -> Result<()> { + self.validate(source, now_ms)?; + if source.session != prepared.catalog.session + || source.log.as_ref().is_some_and(|log| { + log.epoch() != prepared.head.epoch + || log.phase() != crate::node::log_state::NodeLogPhase::Open + }) + { + return Err(Error::Fenced); + } + Ok(()) } } @@ -131,6 +147,50 @@ pub(super) async fn load_catalog( } Ok(catalog) } +/// A Cell pin cannot exist outside the complete canonical inventory, including +/// when callers use the lower-level CellAuthority transition directly. +pub(crate) async fn ensure_enrollment( + layout: &cellule_ltx::CellStorageLayout, + control: &Control, +) -> Result<()> { + let pin = control.bundle_binding.ok_or(Error::Fenced)?; + let (body, _) = layout + .store() + .get_with_etag_bounded( + &layout.node_path(pin.session.as_bytes()), + crate::node::MAX_NODE_BYTES, + ) + .await?; + let NodeRecord::Advertisement(node) = NodeRecord::decode_canonical(&body)? else { + return Err(Error::Fenced); + }; + if node.session != pin.session + || node.log.as_ref().is_some_and(|log| { + log.epoch() != pin.epoch || log.phase() != crate::node::log_state::NodeLogPhase::Open + }) + { + return Err(Error::Fenced); + } + let head = node.bundle.ok_or(Error::PendingPublication)?; + let mut catalog = load_catalog(layout, pin.session, head).await?; + let binding = catalog.binding_mut(pin.digest)?; + if head.epoch != pin.epoch + || binding.phase != BindingPhase::Provisional + || binding.application.as_bytes() != layout.application_id() + || binding.control.bundle_binding != Some(pin) + || binding.control.cell != control.cell + || binding.control.incarnation != control.incarnation + || binding.control.epoch != control.epoch + || binding.control.owner != control.owner + || binding.control.root != control.root + || binding.control.code != control.code + || binding.control.schema != control.schema + { + return Err(Error::Fenced); + } + Ok(()) +} + /// Shared by every lower-level departure CAS. A caller cannot bypass closure /// by using CellAuthority directly instead of the live actor's drain API. pub(crate) async fn ensure_departure( diff --git a/crates/cellule-runtime/src/node/bundle/tests/faults.rs b/crates/cellule-runtime/src/node/bundle/tests/faults.rs index 6a3df959..a6d79874 100644 --- a/crates/cellule-runtime/src/node/bundle/tests/faults.rs +++ b/crates/cellule-runtime/src/node/bundle/tests/faults.rs @@ -30,6 +30,25 @@ impl ObjectStore for ReplyFault { opts: PutOptions, ) -> object_store::Result { let mode = self.mode.load(Ordering::SeqCst); + let node_update = path.as_ref().contains("/nodes/") + && matches!(opts.mode, object_store::PutMode::Update(_)); + if node_update && mode == 4 { + self.mode + .compare_exchange(4, 5, Ordering::SeqCst, Ordering::SeqCst) + .unwrap(); + } else if node_update && mode == 5 { + return Err(denied()); + } + if mode == 3 + && path.as_ref().ends_with("/control.json") + && matches!(opts.mode, object_store::PutMode::Update(_)) + && self + .mode + .compare_exchange(3, 0, Ordering::SeqCst, Ordering::SeqCst) + .is_ok() + { + return Err(denied()); + } let eligible = (mode == 1 && path.as_ref().ends_with(".cnb")) || (mode == 2 && path.as_ref().contains("/nodes/") @@ -77,6 +96,157 @@ impl ObjectStore for ReplyFault { } } +#[tokio::test] +async fn interrupted_pin_cas_retains_a_provisional_inventory_obligation() { + let faults = Arc::new(ReplyFault::default()); + let mut f = Fixture::with_store(faults.clone()).await; + let cell = f.unbound_cell_for_application(4, [9; 16]).await; + faults.mode.store(3, Ordering::SeqCst); + assert!( + f.directory + .bind_bundle_cell(&f.node, &cell.authority, &cell.control, NOW) + .await + .is_err() + ); + let reserved = f + .directory + .load(SessionId::from_bytes([1; 16]), NOW) + .await + .unwrap() + .unwrap(); + let catalog = load_catalog( + &f.layout, + SessionId::from_bytes([1; 16]), + reserved.advertisement().bundle_head().unwrap(), + ) + .await + .unwrap(); + assert_eq!(catalog.bindings.len(), 1); + assert_eq!(catalog.bindings[0].phase, BindingPhase::Provisional); + assert!( + cell.authority + .load(cell.control.value().cell) + .await + .unwrap() + .unwrap() + .value() + .bundle_binding + .is_none() + ); + assert!(matches!( + f.directory.withdraw(&reserved, NOW).await, + Err(Error::PendingPublication) + )); + let (opened, pinned) = f + .directory + .bind_bundle_cell(&reserved, &cell.authority, &cell.control, NOW) + .await + .unwrap(); + assert_eq!( + pinned.value().bundle_binding, + catalog.bindings[0].control.bundle_binding + ); + let catalog = load_catalog( + &f.layout, + SessionId::from_bytes([1; 16]), + opened.advertisement().bundle_head().unwrap(), + ) + .await + .unwrap(); + assert_eq!(catalog.bindings.len(), 1); + assert_eq!(catalog.bindings[0].phase, BindingPhase::Open); +} + +#[tokio::test] +async fn lower_level_pin_cas_cannot_bypass_catalog_reservation() { + let mut f = Fixture::new().await; + let cell = f.unbound_cell_for_application(4, [9; 16]).await; + let mut next = cell.control.value().clone(); + next.revision += 1; + next.progress += 1; + next.bundle_binding = Some(BundleBindingRef { + session: SessionId::from_bytes([1; 16]), + epoch: EPOCH, + digest: Digest::from_bytes([55; 32]), + }); + assert!( + cell.authority + .transition(&cell.control, next, Transition::BindBundle) + .await + .is_err() + ); + assert!( + cell.authority + .load(cell.control.value().cell) + .await + .unwrap() + .unwrap() + .value() + .bundle_binding + .is_none() + ); +} + +#[tokio::test] +async fn interrupted_activation_retains_the_already_selected_cell_pin() { + let faults = Arc::new(ReplyFault::default()); + let mut f = Fixture::with_store(faults.clone()).await; + let cell = f.unbound_cell_for_application(4, [9; 16]).await; + faults.mode.store(4, Ordering::SeqCst); + let error = f + .directory + .bind_bundle_cell(&f.node, &cell.authority, &cell.control, NOW) + .await + .err() + .unwrap(); + let Error::Storage(cellule_store::StorageError::NotSupported { + source: object_store::Error::NotSupported { source }, + }) = error + else { + panic!("original provider error must survive unchanged-head reconciliation: {error:?}"); + }; + assert_eq!(source.to_string(), "injected bundle reply failure"); + let pinned = cell + .authority + .load(cell.control.value().cell) + .await + .unwrap() + .unwrap(); + assert!(pinned.value().bundle_binding.is_some()); + let reserved = f + .directory + .load(SessionId::from_bytes([1; 16]), NOW) + .await + .unwrap() + .unwrap(); + assert!(matches!( + f.directory.withdraw(&reserved, NOW).await, + Err(Error::PendingPublication) + )); + assert!(matches!( + f.directory + .load_bundle_coverage(&cell.authority, &pinned, Limits::default()) + .await, + Err(Error::PendingPublication) + )); + faults.mode.store(0, Ordering::SeqCst); + let (opened, same_pin) = f + .directory + .bind_bundle_cell(&reserved, &cell.authority, &pinned, NOW) + .await + .unwrap(); + assert_eq!(same_pin.value(), pinned.value()); + let catalog = load_catalog( + &f.layout, + SessionId::from_bytes([1; 16]), + opened.advertisement().bundle_head().unwrap(), + ) + .await + .unwrap(); + assert_eq!(catalog.bindings.len(), 1); + assert_eq!(catalog.bindings[0].phase, BindingPhase::Open); +} + #[tokio::test] async fn lost_immutable_reply_grants_no_proof_and_retry_reuses_exact_bytes() { let faults = Arc::new(ReplyFault::default()); diff --git a/crates/cellule-runtime/src/node/bundle/tests/mod.rs b/crates/cellule-runtime/src/node/bundle/tests/mod.rs index 887b9887..97372741 100644 --- a/crates/cellule-runtime/src/node/bundle/tests/mod.rs +++ b/crates/cellule-runtime/src/node/bundle/tests/mod.rs @@ -103,6 +103,17 @@ impl Fixture { self.cell_for_application(byte, [9; 16]).await } async fn cell_for_application(&mut self, byte: u8, application: [u8; 16]) -> Cell { + let mut cell = self.unbound_cell_for_application(byte, application).await; + let (node, control) = self + .directory + .bind_bundle_cell(&self.node, &cell.authority, &cell.control, NOW) + .await + .unwrap(); + self.node = node; + cell.control = control; + cell + } + async fn unbound_cell_for_application(&mut self, byte: u8, application: [u8; 16]) -> Cell { let layout = self.layout.for_application(application); let cell = CellId::from_bytes([byte; 32]); let incarnation = IncarnationId::from_bytes([byte + 10; 16]); @@ -148,17 +159,11 @@ impl Fixture { let authority = CellAuthority::new(layout); authority.retain_root_lineage(&prepared).await.unwrap(); let observed = authority.load(cell).await.unwrap().unwrap(); - let (node, control) = self - .directory - .bind_bundle_cell(&self.node, &authority, &observed, NOW) - .await - .unwrap(); - self.node = node; Cell { db, replica, authority, - control, + control: observed, } } fn append( diff --git a/docs/bundle-coverage-implementation.md b/docs/bundle-coverage-implementation.md index 78267ba4..a07286c5 100644 --- a/docs/bundle-coverage-implementation.md +++ b/docs/bundle-coverage-implementation.md @@ -15,9 +15,11 @@ materialized positions. Its [bundle credit path](https://github.com/denoland/cel rechecks bucket state and epoch after the upload before crediting rows. These are architectural references; this module is independently implemented. -Cellule also has independent Cell ownership epochs. It pins each original -writer in Cell control before enrolling it in the complete node catalog. The -existing canonical node-record CAS selects catalog and range together. A +Cellule also has independent Cell ownership epochs. It first reserves a +provisional catalog binding, pins the original writer in Cell control, then +opens the binding through node CAS. Interrupted enrollment remains an inventory +obligation; a lower-level Cell pin CAS cannot bypass reservation. The existing +canonical node-record CAS selects catalog and range together. A heartbeat preserves the head; a conflicting closure requires a fresh proposal. Uploading an immutable object supplies no coverage proof. @@ -25,7 +27,7 @@ Uploading an immutable object supplies no coverage proof. | API or boundary | Behavior | | --- | --- | -| `NodeDirectory::initialize_bundle_lane` / `bind_bundle_cell` | Establish a boot/epoch catalog and pin an original serving Cell's base, code/schema and writer identity before issuing bundled commands | +| `NodeDirectory::initialize_bundle_lane` / `bind_bundle_cell` | Establish a boot/epoch catalog; reserve a provisional inventory entry, pin the original Cell's base/code/schema/writer, then open it before issuing bundled commands | | `NodeLogShipper::submit_assigned` / `NodeDurability::submit_assigned` | Use the existing bounded native shipping lane and return an opaque witness for every frame in a complete capture | | `prepare_node_bundle` / `select_node_bundle` | Reject missing, overlapping, unassigned or cross-binding ranges; verify origin extents before selecting; reconcile only the exact head under the original lease | | `BundleCoverageProof` | Retain authenticated immutable locators, logical endpoint, SQLite position and original base; retain no capture bodies | @@ -96,7 +98,7 @@ fairness and cancellation/drain ownership are not integrated. ## Verification and remaining delivery -Twenty focused tests cover real managed SQLite captures, canonical object-store +Twenty-three focused tests cover real managed SQLite captures, canonical object-store CAS and materialization: two-Cell shared selection, byte-identical roots, a prior Fleet ACK above the selected prefix, held old proposals across closure, hot siblings and dormant bindings, lost replies, lease loss, partial command groups, @@ -110,13 +112,15 @@ The asynchronous tests select another command while an older root is pending. The older root checkpoints a complete prefix without dropping the hot suffix; the next materializer can continue from that root before or after the catalog checkpoint. Origin lookup also works between the root CAS and catalog CAS. +Enrollment fault tests cover failure before pin selection and after the pin +but before catalog activation; neither grants a proof or clean maintenance. The two-Cell selection test observes **one immutable PUT plus one node CAS**, with both Cell roots unchanged. That excludes enrollment, root materialization, catalog checkpoints, compaction and maintenance and must not be reported as total PUTs/command, TPS or a latency result. -The isolated source snapshot passed all contributor checks: **1,884 workspace +The isolated source snapshot passed all contributor checks: **1,887 workspace tests passed, 38 ignored; 58 local LTX tests passed**. All-feature/all-target checking, warnings-denied Clippy and API docs, formatting, boundaries, module ownership, document syntax/links and SQL/peer validation also passed. Ignored @@ -131,7 +135,8 @@ Remaining work before responses can use bundle proof: lookup/checkpoint index; measure the full checkpoint and collection cost. 3. Fence and seal a failed original node's complete follower-issued suffix, including ACKs above the selected head, before reconstructing and closing - every original binding. Current departure/maintenance guards refuse this + every original binding, including interrupted provisional enrollments. + Current departure/maintenance guards refuse this unfinished recovery rather than advancing a replacement writer. 4. Implement complete cross-Cell reference inventory, pins and grace-qualified collection. Preserve original catalog and base dependencies throughout. diff --git a/docs/bundle-coverage-proof.md b/docs/bundle-coverage-proof.md index 2c31467d..c02258b4 100644 --- a/docs/bundle-coverage-proof.md +++ b/docs/bundle-coverage-proof.md @@ -74,7 +74,7 @@ read node advertisement would reinstate a fencing race. | Object or capability | Exact meaning | | --- | --- | | Cell binding | Cell/incarnation, writer epoch, exact base root/schema/code, permitted node boot/log epoch, and a unique binding identity pinned by Cell control | -| Immutable binding catalog | Complete Open/Closing bindings plus terminal Closed endpoints; closed binding IDs cannot be re-added | +| Immutable binding catalog | Provisional enrollment obligations, complete Open/Closing bindings and terminal Closed endpoints; closed binding IDs cannot be re-added | | Immutable range manifest | Contiguous ordered node ranges, exact Cell bindings and commit/transaction intervals, scoped byte extents and digests, complete native outcome/dependency coverage, predecessor head | | Node selection | Post-upload CAS of both catalog and range head while the original node/log remains Open and every row's binding remains active | | `BundleCoverageProof` | Opaque capability minted only after exact selection reconciliation and complete dependency verification; uploaded bytes cannot construct it | From dd8928e88b7234566e04a24a4f9ec130c0ebab3b Mon Sep 17 00:00:00 2001 From: forhappy Date: Wed, 7 Oct 2026 11:36:28 -0700 Subject: [PATCH 016/102] Document PR 67 write performance reevaluation --- docs/bundle-coverage-implementation.md | 4 +- docs/pr67-performance-reevaluation.md | 143 +++++++++++++++++++++++++ docs/write-performance-delivery.md | 8 ++ 3 files changed, 154 insertions(+), 1 deletion(-) create mode 100644 docs/pr67-performance-reevaluation.md diff --git a/docs/bundle-coverage-implementation.md b/docs/bundle-coverage-implementation.md index a07286c5..cc62a65d 100644 --- a/docs/bundle-coverage-implementation.md +++ b/docs/bundle-coverage-implementation.md @@ -4,7 +4,9 @@ The connected protocol APIs now implement shared selection, independently awaitable root materialization and complete live-writer closure. They are **not enabled in the ordinary actor response path**. The prior performance regression and failed qualification remain the baseline. This slice establishes ordering -and reconstruction evidence; it makes no new throughput or latency claim. +and reconstruction evidence. The [fresh application-path benchmark](pr67-performance-reevaluation.md) +measures `7fc0793`; it does not exercise bundle-based responses or establish +write parity. ## Celld reference and Cellule adaptation diff --git a/docs/pr67-performance-reevaluation.md b/docs/pr67-performance-reevaluation.md new file mode 100644 index 00000000..08778261 --- /dev/null +++ b/docs/pr67-performance-reevaluation.md @@ -0,0 +1,143 @@ +# PR 67 write performance reevaluation + +Measured 2026-10-07. **Write parity remains unqualified.** The latest candidate +improves low-load Fleet latency and modestly improves overloaded bucket results +against main, but completes fewer commands at the Fleet target. These are fresh +measurements of the enabled application path, not a benchmark of the experimental +bundle ACK APIs. + +## Sources and experiment + +| Arm | Pinned revision | +| --- | --- | +| Latest PR 67 candidate | `7fc079345b5214a18a001833732a84039e4485db` | +| Main | `831877cf5af864b11b7bda79b594639fbd233e94` | +| celld v0.6.1 | `f2bf648663a610eefde71f3547ad61e9b896b1f0` | + +Nine serial cases used the same SQL-ledger application: 1,000 uniform Cells, +96-byte values, INSERT plus SELECT and a durable request/result ledger, +128 clients and 128 queued offers. Each case measured 300 seconds after +30 seconds warmup, with fresh provider data. This is one matched repetition per +point, not the proposal's three-repetition qualification or a bounded-KV test. + +The owner, two Fleet followers, client and RustFS shared an ARM64 Linux Docker VM +with 8 CPUs and 16 GiB RAM. Node state used 4-GiB tmpfs mounts. RustFS retained its +2-CPU/2-GiB quota; the client retained its 4-CPU/4-GiB quota. Cellule retained +64 MiB capture admission and a 1-GiB managed disk budget. The Docker data disk +was expanded from 120 to 200 GiB before every arm to preserve inode headroom; +older results are not a controlled before/after pair for that change. + +The latest source was exported into a fresh external directory and compiled in +release mode. Main reused its verified immutable release binary because remote +main was unchanged. Neither arm used a measurement overlay. Client and ACK +auditor binaries, fixture sources, loaded runner and pinned container images +matched byte-for-byte. Latency starts at scheduled arrival; errors, drops, +unissued offers and late completions remain failures. + +## Measured requests + +| Mode / offered writes per second | Main completed TPS / p99 ms | Latest candidate completed TPS / p99 ms | celld completed TPS / p99 ms | +| --- | ---: | ---: | ---: | +| Fleet / 100 | 100.000 / 34.7 | 100.000 / 19.9 | 100.000 / 16.0 | +| Fleet / 15,000 | 545.647 / 2,823.8 | 319.063 / 146.6 | 1,039.493 / 102.0 | +| Bucket / 2,000 | 154.840 / 9,677.6 | 167.233 / 6,299.8 | 635.750 / 1,553.6 | + +The two target rows are **overloaded completion rates, not sustainable capacity**. +Their all-attempt p99 values include failures. In particular, the candidate's +lower Fleet stress p99 cannot establish a latency win with millions of errors. + +| Target case | Measurement request errors | Dropped offers | +| --- | ---: | ---: | +| Main Fleet | 14 | 4,336,039 | +| Candidate Fleet | 3,629,118 | 775,163 | +| celld Fleet | 2,120,374 | 2,067,778 | +| Main bucket | 0 | 553,292 | +| Candidate bucket | 0 | 549,574 | +| celld bucket | 0 | 409,019 | + +All nine cases generated every planned measurement offer. At 100 Fleet writes/s, +the candidate's p99 was 42.7% below main and 24.4% above celld. Each arm's 34,001 +seed/warmup/window ACKs passed every warm and cold GET and exact original retry. +Original nodes were removed before bucket-only restore. + +At the Fleet target, the candidate completed 41.5% fewer commands than main. +At the bucket target, it completed 8.0% more than main with 34.9% lower p99. +Neither observation establishes repeatability, qualified capacity or parity. +Main's fresh low-load latency differs substantially from its historical samples; +changes between sessions cannot be attributed solely to the latest source. + +## Recovery and failure evidence + +| Case | Warm / cold ACK and exact-retry audit | Original fleet drain | +| --- | --- | ---: | +| Main Fleet 100 | All 34,001 pass both | 4.69 s | +| Candidate Fleet 100 | All 34,001 pass both | 4.17 s | +| celld Fleet 100 | All 34,001 pass both | 4.17 s | +| Main Fleet target | All 203,350 pass both | 20.47 s | +| Candidate Fleet target | Warm: 107,065 HTTP 503 errors; cold not reached | No successful drain result | +| celld Fleet target | Warm: 487,194 HTTP 500 errors; cold not reached | No successful drain result | +| Main bucket target | All 53,387 pass both | 5.35 s | +| Candidate bucket target | All 57,416 pass both | 4.16 s | +| celld bucket target | Warm: all 218,423 pass; cold: 111 HTTP 500 errors | 9.79 s | + +The candidate Fleet provider was OOM-killed about four seconds after the final +measurement metric sample. The warm audit therefore ran with an unavailable +provider. Preserve this infrastructure failure separately from Cellule's +publication pressure; the HTTP 503s alone do not identify a framework availability +bug or acknowledged-state loss. The complete case fails infrastructure and +recovery qualification even though the measurement window produced a TPS value. + +The celld Fleet owner reported `database or disk is full` during local WAL +capture. Its warm HTTP 500 failures and missing cold audit leave durability +unverified. The stopped tmpfs could not be inspected afterward; the log is not +a verified byte/inode measurement or evidence of object-provider exhaustion. + +The celld bucket provider remained healthy through cold recovery, but the cold +audit had 111 HTTP 500 errors and only 218,312 of 218,423 exact retries were +checked. Its cold log records a 30-second S3 LIST timeout during restore. This +does not isolate the cause of every HTTP error or demonstrate missing durable +bytes; the unchanged complete audit fails. + +## Publication and architecture findings + +Cost counters measure storage API operations; SDK-internal retry attempts are +not individually counted. + +The candidate Fleet 100 window cost **5.468 successful provider PUTs/command**, +versus **5.493** for main. It selected about one root per command, and 99.1% of +captures used singleton uploads. Enrollment cost was 0.0138 GETs/command across +owner and receivers. The strict debt trend check failed for both Cellule arms; +the candidate's sampled byte slope was +18.55 bytes/s. A small observed backlog +does not override the unchanged qualification gate. + +At the Fleet target, main materialized 4.55 logical commands/selected root, +versus 2.12 for the candidate. Candidate publication averaged 3,620.7 ms and +worker round trip 23.16 ms, versus main's 2,858.8 ms and 10.60 ms. These are +overlapping event populations, not additive command service times. The candidate +charged about 61 MiB of its 64-MiB retained budget by minute two; its minute-four +object frontier stopped advancing while Fleet proofs continued. Complete issued +range tracking does not itself resolve publication or admission pressure. + +Candidate follower durable append averaged 21.82–21.89 ms per batched operation +at the Fleet target. Data sync averaged about 0.001 ms on tmpfs. This does not +establish physical-device fsync performance or a RocksDB comparison. Window PUT +ratios at the Fleet target exclude substantial unpaid publication and trailing +work and must not be credited as a steady-state cost reduction. + +The healthy candidate bucket case cost **3.795 successful PUTs/command**, with +1.024 materialized commands/selected root. Its sampled age trend passed, but +delivery and latency failed. Per-Cell root/lineage/authority selection remains +the cost floor of the enabled path. The experimental shared node selection, +authenticated locators and independent materialization APIs are still absent +from application response/read/retry and background scheduling. See the +[implementation status](bundle-coverage-implementation.md). + +Read-only/mixed capacity, A/A variance, physical-device durability, owner-loss, +grace-qualified collection and complete M0–M5 qualification remain unverified +by this write-only matrix. +The [proposal](write-performance-proposal.md) retains all original acceptance +gates. Source manifests, binaries, request journals, failed cases and machine +reports are retained outside the repository under the `pr67-7fc0793` evidence +tag; only this concise result is committed. The external evidence index is +`pr67-7fc0793-reevaluation-evidence-index.json`, SHA-256 +`1ebb4d3e19f1265afa69047c0237612dea288bee1f8a5789d457413b4c960855`. diff --git a/docs/write-performance-delivery.md b/docs/write-performance-delivery.md index 117990f2..9a253c02 100644 --- a/docs/write-performance-delivery.md +++ b/docs/write-performance-delivery.md @@ -6,6 +6,14 @@ comparison retains exact retry and cold-state audits. **Celld write parity has not been established.** The [proposal](write-performance-proposal.md) remains the acceptance contract; completing tests or a load run does not pass its gates. +The [PR 67 reevaluation](pr67-performance-reevaluation.md) measures the latest +protocol implementation at `7fc0793` in nine fresh matched Docker cases. Fleet +100/s p99 is 19.9 ms versus main's 34.7 ms and celld's 16.0 ms. Target-load +delivery still fails: candidate Fleet completion is below main, bucket is +modestly better, and provider/recovery failures remain explicitly recorded. +The older measurements below are historical and are not measurements of the +latest bundle APIs. + ## Delivered behavior | Change | Measurable result | Preserved contract | From 075b2cd45cb2762f9795aab643e0a6ad14e4ca8d Mon Sep 17 00:00:00 2001 From: forhappy Date: Wed, 7 Oct 2026 13:29:21 -0700 Subject: [PATCH 017/102] Use WAL NORMAL for externally durable runtime activations --- crates/cellule-ltx/docs/README.md | 12 +++- crates/cellule-ltx/src/capture/mod.rs | 10 +++ crates/cellule-ltx/src/db/mod.rs | 27 ++++++++ crates/cellule-ltx/src/db/tests/mod.rs | 63 +++++++++++++++++++ crates/cellule-runtime/docs/runtime.md | 7 +++ .../docs/vfs-ltx-scale-plan.md | 2 +- .../cellule-runtime/src/cell/executor/mod.rs | 12 ++++ .../src/cell/executor/tests.rs | 56 +++++++++++++++++ .../runtime/lifecycle/durability/proofs.rs | 8 +++ docs/performance-plan.md | 6 +- 10 files changed, 199 insertions(+), 4 deletions(-) diff --git a/crates/cellule-ltx/docs/README.md b/crates/cellule-ltx/docs/README.md index 9cf7c488..d5a1d7e3 100644 --- a/crates/cellule-ltx/docs/README.md +++ b/crates/cellule-ltx/docs/README.md @@ -473,7 +473,8 @@ write. - **First capture.** The first capture creates a WAL frame only when needed, then binds the inherited checksum state to that WAL's complete committed prefix. - **Later captures.** Later captures retain the usual salt and committed-boundary - checks; SQLite `synchronous=FULL` and resume verification are unchanged. + checks. Standalone writers retain SQLite `synchronous=FULL`; runtime activations + explicitly select `Db::use_external_durability`. Resume verification is unchanged. ## Local durability boundaries @@ -481,6 +482,7 @@ write. | Operation | Local barrier | What it proves | May release a Cell response? | | --- | --- | --- | --- | | SQLite commit | SQLite WAL sync under `synchronous=FULL` | The local commit reached SQLite's WAL boundary | No | +| SQLite commit after `use_external_durability()` | SQLite WAL `NORMAL`; most commits omit the WAL sync | The commit is locally readable, but can be lost after OS crash or power loss | No | | `capture()` | LTX file sync, rename, parent sync, then first-cut directory-chain sync | The returned standalone LTX cuts have durable bytes and names | No | | `capture_deferred()` | No LTX file or name barrier; a sparse writer updates its mutable checksum sidecar without syncing it | The cut is readable for publication, but its LTX durability is pending | No | | `durability_barrier()` | Pending LTX files, their parent directories, and the first-cut directory chain | Those deferred local cuts are durable | No | @@ -492,6 +494,14 @@ missing or invalid sidecar discards warm reuse; the runtime restores its authority-pinned root. Process-kill tests do not prove physical power-loss durability for SQLite, the LTX file, or the filesystem's sync implementation. +`use_external_durability()` is an explicit session contract for callers that +release responses and reads only after an exact external proof. It configures +the application writer and capture's maintenance writer before mutations; +the read-mark connection never writes. It refuses pending captures and fences +partial configuration failures. SQLite still synchronizes checkpoint backfill +under `NORMAL`. The runtime selects this mode during bootstrap and after exact +restored-image validation; direct `Db` and `CellReplica` opens default to `FULL`. + ## Preparing a Cell root diff --git a/crates/cellule-ltx/src/capture/mod.rs b/crates/cellule-ltx/src/capture/mod.rs index 299863c0..97587d71 100644 --- a/crates/cellule-ltx/src/capture/mod.rs +++ b/crates/cellule-ltx/src/capture/mod.rs @@ -272,6 +272,16 @@ impl CaptureEngine { } } + pub(crate) fn use_external_durability(&self) -> Result<()> { + // Checkpoint maintenance also commits control-table writes. It uses + // the same externally proven boundary as the application writer; the + // retained read-mark connection never writes. NORMAL still synchronizes + // WAL and database backfill at SQLite's checkpoint boundaries. + self.conn + .pragma_update(None, "synchronous", "NORMAL") + .map_err(LtxError::from) + } + pub fn wal_path(&self) -> PathBuf { let mut s = self.path.clone().into_os_string(); s.push("-wal"); diff --git a/crates/cellule-ltx/src/db/mod.rs b/crates/cellule-ltx/src/db/mod.rs index d87438fb..b070c1fe 100644 --- a/crates/cellule-ltx/src/db/mod.rs +++ b/crates/cellule-ltx/src/db/mod.rs @@ -65,6 +65,33 @@ impl Db { self.writer.get_interrupt_handle() } + /// Uses SQLite WAL `NORMAL` when external proofs own command durability. + /// + /// Call before accepting mutations in a fresh or exactly restored session. + /// A successful local commit can be lost after an OS crash or power loss; + /// the caller must await an exact object or recoverable follower proof before + /// acknowledging or exposing it, and must never recover from unverified local + /// residue. Standalone opens retain `FULL` unless this is explicitly selected. + /// LTX capture and durability barriers retain their existing contracts. + /// A partial configuration failure fences the session. + pub fn use_external_durability(&mut self) -> Result<()> { + self.ensure_active()?; + if self.has_pending_capture() || self.retained_segments != 0 { + return Err(LtxError::InvalidState( + "external durability requires a session without pending captures", + )); + } + let configured = self + .writer + .pragma_update(None, "synchronous", "NORMAL") + .map_err(LtxError::from) + .and_then(|()| self.capture.use_external_durability()); + if configured.is_err() { + self.fenced = true; + } + configured + } + #[cfg(feature = "replica")] pub(crate) fn open_cell_paged( database: crate::CellWritableDatabase, diff --git a/crates/cellule-ltx/src/db/tests/mod.rs b/crates/cellule-ltx/src/db/tests/mod.rs index 4cf46dda..e4dbaa87 100644 --- a/crates/cellule-ltx/src/db/tests/mod.rs +++ b/crates/cellule-ltx/src/db/tests/mod.rs @@ -51,6 +51,69 @@ fn managed_connections_set_the_budgeted_page_cache() { assert_eq!(cache_kib, -MANAGED_CONNECTION_PAGE_CACHE_KIB); } +#[test] +fn external_durability_is_explicit_and_refuses_an_uncaptured_commit() { + let temp = tempfile::TempDir::new().unwrap(); + let mut db = Db::open(&temp.path().join("durability.sqlite"), Limits::default()).unwrap(); + let synchronous = |db: &mut Db| { + db.query_with(|connection| { + connection.query_row("PRAGMA synchronous", [], |row| row.get::<_, i64>(0)) + }) + .unwrap() + }; + assert_eq!(synchronous(&mut db), 2); + db.transaction(|tx| tx.execute_batch("CREATE TABLE witness(value INTEGER)")) + .unwrap(); + assert!(matches!( + db.use_external_durability(), + Err(LtxError::InvalidState(_)) + )); + assert_eq!(synchronous(&mut db), 2); + db.capture().unwrap(); + assert!(db.use_external_durability().is_err()); + db.close().unwrap(); + + let mut external = Db::open(&temp.path().join("external.sqlite"), Limits::default()).unwrap(); + external.use_external_durability().unwrap(); + assert_eq!(synchronous(&mut external), 1); + external + .transaction(|tx| { + tx.execute_batch("CREATE TABLE witness(value INTEGER); INSERT INTO witness VALUES(7)") + }) + .unwrap(); + let cuts = external.capture().unwrap(); + let restored = temp.path().join("restored.sqlite"); + let plan = VerifiedPlan::new(&cuts.segments, cuts.position, Limits::default()).unwrap(); + restore_exact(&plan, &restored).unwrap(); + let connection = Connection::open(restored).unwrap(); + assert_eq!( + connection + .query_row("SELECT value FROM witness", [], |row| row.get::<_, i64>(0)) + .unwrap(), + 7 + ); + external.close().unwrap(); +} + +#[test] +fn external_durability_configuration_failure_fences_the_session() { + let temp = tempfile::TempDir::new().unwrap(); + let mut db = Db::open(&temp.path().join("configuration.sqlite"), Limits::default()).unwrap(); + // SQLite refuses a safety-level change inside a transaction. The public + // transaction callback cannot leave one behind; inject it at this seam to + // check the configuration error path, including refusal of subsequent SQL. + db.writer.execute_batch("BEGIN").unwrap(); + assert!(matches!( + db.use_external_durability(), + Err(LtxError::Sqlite(_)) + )); + assert!(matches!(db.transaction(|_| Ok(())), Err(LtxError::Fenced))); + assert!(matches!( + db.use_external_durability(), + Err(LtxError::Fenced) + )); +} + #[test] fn capture_reports_deterministic_bounded_timing_for_real_ltx_work() { let temp = tempfile::TempDir::new().unwrap(); diff --git a/crates/cellule-runtime/docs/runtime.md b/crates/cellule-runtime/docs/runtime.md index 36fcecbd..9a5c2970 100644 --- a/crates/cellule-runtime/docs/runtime.md +++ b/crates/cellule-runtime/docs/runtime.md @@ -356,6 +356,13 @@ A Cell can install `primitives::capacity::SCHEMA` and use `CommandContext::reser ## Publish before replying +Runtime bootstrap and verified restored activations select SQLite WAL +`synchronous=NORMAL` through `Db::use_external_durability`. Local commit success +does not release a response or query: the exact object-root or recoverable +follower proof remains the durability boundary. Unverified local WAL and +capture residue cannot authorize recovery or warm reuse. Standalone LTX opens +and direct `CellExecutor::new` callers retain their supplied database mode. + `PendingCommit` owns the request identity, predecessor control, encoded reply, commit sequence, and captured cuts. The actor doesn't accept the next mutation until this commit reaches a terminal publication result. ```text diff --git a/crates/cellule-runtime/docs/vfs-ltx-scale-plan.md b/crates/cellule-runtime/docs/vfs-ltx-scale-plan.md index 32e1c9c5..0a44d13a 100644 --- a/crates/cellule-runtime/docs/vfs-ltx-scale-plan.md +++ b/crates/cellule-runtime/docs/vfs-ltx-scale-plan.md @@ -468,7 +468,7 @@ workloads before and after each change: | 2 | Root preparation reports one duration and upload count. It does not distinguish predecessor root GET/HEAD, directory reads, immutable PUT, provider wait, and authority CAS. | Extend the RustFS cost harness or an instrumented store to count GET, HEAD, PUT, bytes, attempts, and time by finite phase. Run single and many-Cell concurrency, plus hot-Cell increasing offered load. Report publisher queue age, unpublished bytes, admission rejections, root lag, provider saturation, and sustainable published roots/s. A latency change passes only if throughput or p95/p99 improves without increasing failed proofs or provider pressure. | | 3 | A cached predecessor still needs origin presence checks. `load_graph` HEAD-checks cached root metadata; a test requires the next prepare to fail when those objects disappear. | Measure the HEAD wave separately. Keep the [missing-metadata invariant](../../cellule-ltx/tests/cell/roots/lifecycle.rs) while testing any reuse of verified predecessor state. Do not remove HEADs solely because metadata is cached or because PUT keys are content addressed. Compare chain lengths around 1, 96, and 97 descriptors before considering reuse of unchanged segment pages. | | 4 | Tiny steady writes hide checkpoint, full-image fallback, and sparse activation tails. A capture can read the whole WAL or emit a full database image, while the shared paged I/O driver has a bounded request queue and jobs. | Run hot and skewed Cells across checkpoint and compaction thresholds, a pinned reader, large changed-page sets, and simultaneous cold opens. Attribute checkpoint runs/busy/restarts, full WAL reads, full-image bytes, page-fault queue and deadline errors, cache misses, provider I/O, and p99. Test with the 1 GiB node limit; retain exact-root and disk-admission fault tests. | -| 5 | SQLite uses `synchronous=FULL` on managed connections even though the Cell response waits for an external proof. The local sync might be visible in write latency, but no power-loss equivalence has been established. | Benchmark SQLite commit and response latency with the current mode as control. Consider a different mode only in an isolated experiment that proves no response can use an unverified local WAL or continuation after power loss, including failure before capture, after follower proof, and during object publication. Keep the current mode until crash and recovery qualification justifies a contract change. | +| 5 | Runtime bootstrap and verified restored activation explicitly use `synchronous=NORMAL`; standalone LTX writers retain `FULL`. Responses still require an exact external proof, and partial mode configuration fences the session. | Compare against the unchanged FULL candidate with matched tmpfs and persistent-storage diagnostics. Preserve capture-failure fencing, lost-response reconciliation, all-ACK cold recovery, exact retry and verified continuation checks. Process and Docker tests do not qualify physical power-loss durability; retain that qualification gap. | Response-winner instrumentation is now wired through the final command/effect reply boundary and exported as `crab_cell_command_responses_total`, diff --git a/crates/cellule-runtime/src/cell/executor/mod.rs b/crates/cellule-runtime/src/cell/executor/mod.rs index c78e5b30..3936a73a 100644 --- a/crates/cellule-runtime/src/cell/executor/mod.rs +++ b/crates/cellule-runtime/src/cell/executor/mod.rs @@ -355,6 +355,12 @@ impl CellExecutor { schema: u32, initialize: impl FnOnce(&cellule_ltx::rusqlite::Transaction<'_>) -> Result<()>, ) -> Result<(Self, CaptureBatch, Option)> { + // Bootstrap and every later response wait for an external proof. Local + // WAL sync would duplicate that boundary; failed sessions are discarded. + if let Err(error) = db.use_external_durability() { + let _ = db.close(); + return Err(error.into()); + } let initialized = db.transaction_with(|transaction| { crate::cell::schema::install_runtime_schema_in(transaction, cell, incarnation, schema)?; initialize(transaction)?; @@ -452,6 +458,12 @@ impl CellExecutor { let _ = db.close(); return Err(error); } + // Validate the exact authoritative image before enabling the replicated + // writer. This includes sparse, cold-restored and verified warm resumes. + if let Err(error) = db.use_external_durability() { + let _ = db.close(); + return Err(error.into()); + } let mut executor = Self::new(db, cell, incarnation, schema); executor.published_sequence = root.commit_sequence; Ok(executor) diff --git a/crates/cellule-runtime/src/cell/executor/tests.rs b/crates/cellule-runtime/src/cell/executor/tests.rs index bbb66822..8039e401 100644 --- a/crates/cellule-runtime/src/cell/executor/tests.rs +++ b/crates/cellule-runtime/src/cell/executor/tests.rs @@ -60,6 +60,62 @@ fn code_only_registry() -> crate::Registry { registry.finish().unwrap() } +#[tokio::test] +async fn externally_durable_activation_uses_normal_before_sql_and_after_restore() { + let directory = tempfile::TempDir::new().unwrap(); + let cell = CellId::from_bytes([211; 32]); + let incarnation = IncarnationId::from_bytes([212; 16]); + let replica = cellule_ltx::CellReplica::new( + cellule_ltx::CellStorageLayout::new( + cellule_store::Store::new(std::sync::Arc::new(object_store::memory::InMemory::new())), + object_store::path::Path::from("external-durability"), + [213; 16], + ), + *cell.as_bytes(), + *incarnation.as_bytes(), + cellule_ltx::Limits::default(), + ) + .unwrap(); + let db = replica + .open_new(&directory.path().join("initial.sqlite")) + .unwrap(); + let (mut executor, cuts, _) = CellExecutor::bootstrap(db, cell, incarnation, 1, |tx| { + let synchronous: i64 = tx.query_row("PRAGMA synchronous", [], |row| row.get(0))?; + assert_eq!(synchronous, 1); + tx.execute_batch("CREATE TABLE witness(value INTEGER); INSERT INTO witness VALUES(7)")?; + Ok(()) + }) + .unwrap(); + let root = replica.prepare(None, &cuts, 0, 1).await.unwrap().root(); + executor.confirm_bootstrap_published(&cuts).unwrap(); + executor.close().unwrap(); + let restored = directory.path().join("restored.sqlite"); + let db = replica + .open_root(&root) + .await + .unwrap() + .paged() + .prepare_writable(&restored) + .await + .unwrap() + .open_writable(&restored) + .unwrap(); + let mut executor = CellExecutor::from_restored(db, cell, incarnation, 1, root).unwrap(); + executor + .db + .query_with(|connection| { + let synchronous: i64 = + connection.query_row("PRAGMA synchronous", [], |row| row.get(0))?; + let value: i64 = + connection.query_row("SELECT value FROM witness", [], |row| row.get(0))?; + assert_eq!(synchronous, 1); + assert_eq!(value, 7); + Ok::<_, cellule_ltx::rusqlite::Error>(()) + }) + .unwrap(); + executor.close().unwrap(); +} + #[test] fn full_pending_publication_budget_refuses_new_commands() { let directory = tempfile::TempDir::new().unwrap(); diff --git a/crates/cellule-runtime/tests/runtime/lifecycle/durability/proofs.rs b/crates/cellule-runtime/tests/runtime/lifecycle/durability/proofs.rs index 544f21e7..8e5cb0fc 100644 --- a/crates/cellule-runtime/tests/runtime/lifecycle/durability/proofs.rs +++ b/crates/cellule-runtime/tests/runtime/lifecycle/durability/proofs.rs @@ -23,6 +23,10 @@ async fn accepted_control_cas_with_lost_response_releases_once() { let first = handle .execute(request, digest, 20, 1_024, 1_024, move |transaction| { + assert_eq!( + transaction.query_row("PRAGMA synchronous", [], |row| row.get::<_, i64>(0))?, + 1 + ); observed.fetch_add(1, Ordering::SeqCst); transaction.execute("UPDATE counter SET value = value + 1", [])?; Ok(HandlerOutcome::Success(b"published".to_vec())) @@ -126,6 +130,10 @@ async fn capture_failure_after_sql_commit_fences_until_authoritative_recovery() filesystem.fail_next_capture(); let first = handle .execute(request, digest, 20, 1_024, 1_024, move |transaction| { + assert_eq!( + transaction.query_row("PRAGMA synchronous", [], |row| row.get::<_, i64>(0))?, + 1 + ); observed.fetch_add(1, Ordering::SeqCst); transaction.execute("UPDATE counter SET value = value + 1", [])?; Ok(HandlerOutcome::Success(b"unpublished".to_vec())) diff --git a/docs/performance-plan.md b/docs/performance-plan.md index 2a1087f6..1a1e67cc 100644 --- a/docs/performance-plan.md +++ b/docs/performance-plan.md @@ -84,8 +84,10 @@ The database budget already pays for more connections than the runtime opens: `MANAGED_SQLITE_CONNECTIONS = 3` is charged per Cell in the resource ledger (`crates/cellule-ltx/src/db/mod.rs:16`, `crates/cellule-runtime/src/cell/worker/mod.rs:95`) while `Db` holds a single -writer connection, set to `synchronous=FULL`, `wal_autocheckpoint=0`, and a -64 KiB page cache (`crates/cellule-ltx/src/db/mod.rs:367`). +writer connection, opened with `synchronous=FULL`, `wal_autocheckpoint=0`, and a +64 KiB page cache. Runtime bootstrap and verified restored activation explicitly +select `NORMAL` before accepting mutations; standalone opens retain `FULL`. +The response still waits for an exact external durability proof. A demand fault parks that thread. `Io::page` takes a `std::sync::Mutex` gate and then waits on a `sync_channel` with `recv_timeout` From 226765691c7db792a9e920f475f21897e5ada71a Mon Sep 17 00:00:00 2001 From: forhappy Date: Wed, 7 Oct 2026 16:16:18 -0700 Subject: [PATCH 018/102] Reuse canonical packs for bounded recovery materialization --- crates/cellule-ltx/docs/publication.md | 8 ++ crates/cellule-ltx/src/replica/coalesce.rs | 19 +-- crates/cellule-ltx/src/replica/prepare.rs | 61 +++++++++- .../cellule-ltx/tests/cell/roots/lifecycle.rs | 112 +++++++++++++++++- 4 files changed, 186 insertions(+), 14 deletions(-) diff --git a/crates/cellule-ltx/docs/publication.md b/crates/cellule-ltx/docs/publication.md index 821e9e69..ee327928 100644 --- a/crates/cellule-ltx/docs/publication.md +++ b/crates/cellule-ltx/docs/publication.md @@ -22,6 +22,7 @@ sequenceDiagram | `admit_root_preparation` | Waits for the existing dirty reservation before capture selection; the scoped clone starts no work and grants no authority. | | `CellReplica::prepare` | Verifies cuts and writes immutable root dependencies. | | `prepare_bundle` | Selects this Cell's exact rows from a shared bundle. | +| `prepare_recovered_overlay` | Verifies the exact predecessor and final position; small independent tails use the canonical coalescer and native pack. | | `prepare_compaction` | Rewrites representation without changing logical state. | | `prepare_after_compaction` | Appends to a private compaction while retaining its original authority predecessor. | | `try_admit_scheduled_compaction` | Returns a scoped clone with existing dirty/recovery admission, or defers without waiting behind a queued cohort. | @@ -76,6 +77,13 @@ outside the existing capture bound retain the file-backed upload path. All dependencies still finish before a proposal returns; original follower receipts and the authority CAS continue to govern acknowledgement. +Independent recovery uses the same path when all selected rows, their indexes +and pack headers together fit the existing 256 KiB single-PUT budget. Every +original row is verified before coalescing. If a later row exceeds that budget, +earlier frozen bodies are released and the entire tail retains its ordinary +bundle representation. `prepare_bundle` preserves shared bundle references. +No root format, authority rule or host resource ceiling changes. + A representation-only compaction can remain private while its successor append uploads. `prepare_after_compaction` verifies that the compaction preserves the predecessor's position, commit sequence, Cell and incarnation. The runtime selects diff --git a/crates/cellule-ltx/src/replica/coalesce.rs b/crates/cellule-ltx/src/replica/coalesce.rs index b1ddc31f..eb7041bf 100644 --- a/crates/cellule-ltx/src/replica/coalesce.rs +++ b/crates/cellule-ltx/src/replica/coalesce.rs @@ -15,7 +15,7 @@ pub(super) async fn run( ) -> Result> { if segments.len() < 2 || segments.iter().any(|segment| { - !matches!(segment.body, AppendBody::Native(_)) + !matches!(segment.body, AppendBody::Native(_) | AppendBody::Frozen(_)) || segment.descriptor.info.size_bytes > SINGLE_PUT_BYTES || segment.index.len() as u64 > SINGLE_PUT_BYTES }) @@ -29,18 +29,23 @@ pub(super) async fn run( .ok_or(LtxError::LTXCorrupted) })?; for segment in &segments { - let AppendBody::Native(source) = &segment.body else { - return Err(LtxError::LTXCorrupted); - }; let info = segment.descriptor.info.clone(); - let source = source.clone(); + let body = segment.body.clone(); let index = segment.index.clone(); // One admitted job owns the pinned read and verification. Synchronous // access avoids nesting file jobs in the same bounded pool, and the - // dispatched closure retains the source and admission on cancellation. + // dispatched closure retains the pinned or frozen source and admission + // on cancellation. Both representations pass the same byte/index checks. let merged = replica .host - .run(move || state.apply(source.read_small(info.size_bytes)?, &info, &index)) + .run(move || { + let bytes = match body { + AppendBody::Native(source) => source.read_small(info.size_bytes)?, + AppendBody::Frozen(bytes) => bytes, + _ => return Err(LtxError::LTXCorrupted), + }; + state.apply(bytes, &info, &index) + }) .await .map_err(pinned_storage_error)??; let Some(merged) = merged else { diff --git a/crates/cellule-ltx/src/replica/prepare.rs b/crates/cellule-ltx/src/replica/prepare.rs index 11039704..b84ca7e5 100644 --- a/crates/cellule-ltx/src/replica/prepare.rs +++ b/crates/cellule-ltx/src/replica/prepare.rs @@ -6,6 +6,12 @@ use super::*; +#[derive(Clone, Copy, PartialEq, Eq)] +enum BundleUse { + Shared, + IndependentRecovery, +} + impl CellReplica { /// Appends captured cuts to a private representation-only compaction. /// @@ -214,7 +220,7 @@ impl CellReplica { let mut replica = self.clone(); replica.host = self.host.for_dirty().await?; replica - .prepare_bundle_admitted(base, bundle, commit_sequence, schema) + .prepare_bundle_admitted(base, bundle, commit_sequence, schema, BundleUse::Shared) .await } @@ -223,6 +229,8 @@ impl CellReplica { /// Recovery policy and ownership remain caller-owned. This method accepts /// only this replica's Cell/incarnation rows, requires the declared final /// position to match the bundle, and reuses normal root preparation. + /// Independent small tails use canonical native coalescing and packing; + /// larger tails keep the bundle representation under the existing bounds. pub async fn prepare_recovered_overlay( &self, overlay: &RecoveryOverlay, @@ -245,12 +253,15 @@ impl CellReplica { if final_position != overlay.final_position { return Err(LtxError::ChecksumMismatch); } - let prepared = self - .prepare_bundle( + let mut replica = self.clone(); + replica.host = self.host.for_dirty().await?; + let prepared = replica + .prepare_bundle_admitted( Some(&overlay.predecessor), &overlay.bundle, overlay.final_commit_sequence, schema, + BundleUse::IndependentRecovery, ) .await?; if prepared.root().position != overlay.final_position { @@ -265,6 +276,7 @@ impl CellReplica { bundle: &crate::bundle::Bundle, commit_sequence: u64, schema: u32, + usage: BundleUse, ) -> Result { self.validate_metadata(commit_sequence, schema)?; if bundle.len() > self.limits.max_plan_bytes { @@ -278,8 +290,10 @@ impl CellReplica { let (repository, epoch) = crate::bundle::cell_identity(&self.cell, &self.incarnation); let bundle_digest = bundle.digest(); - let mut inputs = Vec::new(); + let mut inputs: Vec = Vec::new(); let mut selected_bytes = 0_u64; + let mut independent = usage == BundleUse::IndependentRecovery; + let mut independent_bytes = 0_u64; let mut prospective = base_graph .as_ref() .map(|graph| graph.descriptors.clone()) @@ -313,6 +327,25 @@ impl CellReplica { } self.admit_segment_representation(&row.info, pages.len() * crate::paged::ENTRY_BYTES)?; let index_bytes = Bytes::from(crate::paged::encode_index_from_pages(&pages)?); + // Retain at most one canonical small-pack budget of verified native + // inputs. If a later row exceeds it, release every earlier frozen + // body and keep the shared bundle representation for the whole tail. + if independent { + match independent_bytes + .checked_add(packed::HEADER_BYTES) + .and_then(|size| size.checked_add(row.info.size_bytes)) + .and_then(|size| size.checked_add(index_bytes.len() as u64)) + .filter(|size| *size <= upload::SINGLE_PUT_BYTES) + { + Some(size) => independent_bytes = size, + None => { + independent = false; + for input in &mut inputs { + input.body = AppendBody::Bundle; + } + } + } + } inputs.push(AppendInput { info: row.info.clone(), location: BodyLocation::Bundle { @@ -320,7 +353,11 @@ impl CellReplica { offset: row.offset, }, index: index_bytes, - body: AppendBody::Bundle, + body: if independent { + AppendBody::Frozen(bytes) + } else { + AppendBody::Bundle + }, }); } let target = inputs @@ -328,6 +365,18 @@ impl CellReplica { .map(|input| input.info.position()) .ok_or(LtxError::TxNotAvailable)?; self.validate_chain(&prospective, target)?; + if independent { + // Independent recovery already verified every original cut. The + // canonical coalescer and native pack factory can now reduce their + // representation without retaining shared-bundle dependencies. + for input in &mut inputs { + input.location = BodyLocation::Native; + } + } + let retained_bundle = inputs + .iter() + .any(|input| matches!(input.body, AppendBody::Bundle)) + .then_some(bundle); self.prepare_append( base, base_graph.map(AppendBaseState::from), @@ -335,7 +384,7 @@ impl CellReplica { target, commit_sequence, schema, - Some(bundle), + retained_bundle, ) .await } diff --git a/crates/cellule-ltx/tests/cell/roots/lifecycle.rs b/crates/cellule-ltx/tests/cell/roots/lifecycle.rs index 64a23f65..45424180 100644 --- a/crates/cellule-ltx/tests/cell/roots/lifecycle.rs +++ b/crates/cellule-ltx/tests/cell/roots/lifecycle.rs @@ -706,7 +706,10 @@ async fn recovered_overlay_requires_exact_predecessor_and_final_position() { let first = writer.capture().unwrap(); let cell = [81; 32]; let incarnation = [82; 16]; - let replica = replica(Store::new(Arc::new(InMemory::new())), cell, incarnation); + let counted = Arc::new(cellule_store::test_support::CountingObjectStore::new( + Arc::new(InMemory::new()), + )); + let replica = replica(Store::new(counted.clone()), cell, incarnation); let base = replica.prepare(None, &first, 1, 6).await.unwrap().root(); writer @@ -730,6 +733,7 @@ async fn recovered_overlay_requires_exact_predecessor_and_final_position() { .collect(); let bundle = Bundle::encode(entries, Limits::default()).unwrap(); let overlay = RecoveryOverlay::new(base, bundle, tail.position, 2); + counted.reset(); let recovered = replica .prepare_recovered_overlay(&overlay, 6) .await @@ -738,6 +742,30 @@ async fn recovered_overlay_requires_exact_predecessor_and_final_position() { assert_eq!(recovered.predecessor(), Some(base)); assert_eq!(recovered.root().position, tail.position); assert_eq!(recovered.root().commit_sequence, 2); + assert_eq!(counted.put_requests(), 2, "one native pack and one root"); + let direct = replica.prepare(Some(&base), &tail, 2, 6).await.unwrap(); + assert_eq!( + recovered.root(), + direct.root(), + "recovery uses the exact canonical pack factory" + ); + let restored = directory.path().join("recovered.sqlite"); + replica + .open_root(&recovered.root()) + .await + .unwrap() + .restore(&restored) + .await + .unwrap(); + let db = rusqlite::Connection::open(restored).unwrap(); + let events: Vec = db + .prepare("SELECT body FROM events ORDER BY id") + .unwrap() + .query_map([], |row| row.get(0)) + .unwrap() + .collect::>() + .unwrap(); + assert_eq!(events, ["published", "fleet-only"]); let invalid = RecoveryOverlay::new( RootRef { @@ -770,3 +798,85 @@ async fn recovered_overlay_requires_exact_predecessor_and_final_position() { ); writer.close().unwrap(); } + +#[tokio::test] +async fn recovered_overlay_keeps_the_bundle_when_a_later_row_exceeds_the_pack_budget() { + let directory = tempfile::TempDir::new().unwrap(); + let mut writer = + Db::open(&directory.path().join("fallback.sqlite"), Limits::default()).unwrap(); + writer.transaction(|tx| tx.execute_batch("CREATE TABLE payloads(id INTEGER PRIMARY KEY, value BLOB); INSERT INTO payloads VALUES(0, X'00')")).unwrap(); + let cell = [181; 32]; + let incarnation = [182; 16]; + let replica = replica(Store::new(Arc::new(InMemory::new())), cell, incarnation); + let base = replica + .prepare(None, &writer.capture().unwrap(), 1, 1) + .await + .unwrap() + .root(); + writer + .transaction(|tx| tx.execute("INSERT INTO payloads VALUES(1, X'01')", [])) + .unwrap(); + let first = writer.capture().unwrap(); + let payload: Vec = (0_u64..10_000) + .flat_map(|number| *blake3::hash(&number.to_le_bytes()).as_bytes()) + .collect(); + writer + .transaction(|tx| tx.execute("INSERT INTO payloads VALUES(2, ?1)", [&payload])) + .unwrap(); + let last = writer.capture().unwrap(); + assert!( + last.segments + .iter() + .map(|segment| segment.info().size_bytes) + .sum::() + > 256 << 10 + ); + let bundle = Bundle::encode( + first + .segments + .iter() + .chain(&last.segments) + .map(|segment| { + BundleEntry::for_cell( + cell, + incarnation, + segment.info().clone(), + std::fs::read(segment.path()).unwrap(), + ) + }) + .collect(), + Limits::default(), + ) + .unwrap(); + let overlay = RecoveryOverlay::new(base, bundle, last.position, 3); + let recovered = replica + .prepare_recovered_overlay(&overlay, 1) + .await + .unwrap(); + let objects = replica.reachable_objects(&recovered.root()).await.unwrap(); + assert_eq!( + objects + .iter() + .filter(|object| object.kind == CellObjectKind::Bundle) + .count(), + 1 + ); + let restored = directory.path().join("fallback-restored.sqlite"); + replica + .open_root(&recovered.root()) + .await + .unwrap() + .restore(&restored) + .await + .unwrap(); + let db = rusqlite::Connection::open(restored).unwrap(); + let values: Vec> = db + .prepare("SELECT value FROM payloads ORDER BY id") + .unwrap() + .query_map([], |row| row.get(0)) + .unwrap() + .collect::>() + .unwrap(); + assert_eq!(values, vec![vec![0], vec![1], payload]); + writer.close().unwrap(); +} From ad665908509d6c12dce11bc425e5f1d4a4b5c77d Mon Sep 17 00:00:00 2001 From: forhappy Date: Wed, 7 Oct 2026 16:16:23 -0700 Subject: [PATCH 019/102] Index shared coverage and checkpoint exact root cohorts --- .../src/node/bundle/binding.rs | 17 +- .../src/node/bundle/closure.rs | 208 +++++++++--- .../cellule-runtime/src/node/bundle/codec.rs | 25 +- .../src/node/bundle/index/codec.rs | 179 +++++++++++ .../src/node/bundle/index/io.rs | 162 ++++++++++ .../src/node/bundle/index/mod.rs | 300 ++++++++++++++++++ .../src/node/bundle/index/tests.rs | 70 ++++ crates/cellule-runtime/src/node/bundle/mod.rs | 16 +- .../cellule-runtime/src/node/bundle/proof.rs | 8 +- .../src/node/bundle/selection.rs | 30 +- .../cellule-runtime/src/node/bundle/store.rs | 54 ++-- .../src/node/bundle/tests/index/bootstrap.rs | 87 +++++ .../src/node/bundle/tests/index/checkpoint.rs | 133 ++++++++ .../src/node/bundle/tests/index/cohort.rs | 176 ++++++++++ .../node/bundle/tests/index/compatibility.rs | 172 ++++++++++ .../node/bundle/tests/index/copy_on_write.rs | 57 ++++ .../src/node/bundle/tests/index/inventory.rs | 63 ++++ .../src/node/bundle/tests/index/mod.rs | 46 +++ .../src/node/bundle/tests/lifecycle.rs | 25 +- .../src/node/bundle/tests/mod.rs | 22 +- .../src/node/bundle/tests/ranges.rs | 29 ++ docs/bundle-coverage-implementation.md | 101 ++++-- docs/bundle-coverage-proof.md | 23 ++ docs/pr67-normal-wal-reevaluation.md | 85 +++++ docs/write-performance-delivery.md | 11 +- 25 files changed, 1992 insertions(+), 107 deletions(-) create mode 100644 crates/cellule-runtime/src/node/bundle/index/codec.rs create mode 100644 crates/cellule-runtime/src/node/bundle/index/io.rs create mode 100644 crates/cellule-runtime/src/node/bundle/index/mod.rs create mode 100644 crates/cellule-runtime/src/node/bundle/index/tests.rs create mode 100644 crates/cellule-runtime/src/node/bundle/tests/index/bootstrap.rs create mode 100644 crates/cellule-runtime/src/node/bundle/tests/index/checkpoint.rs create mode 100644 crates/cellule-runtime/src/node/bundle/tests/index/cohort.rs create mode 100644 crates/cellule-runtime/src/node/bundle/tests/index/compatibility.rs create mode 100644 crates/cellule-runtime/src/node/bundle/tests/index/copy_on_write.rs create mode 100644 crates/cellule-runtime/src/node/bundle/tests/index/inventory.rs create mode 100644 crates/cellule-runtime/src/node/bundle/tests/index/mod.rs create mode 100644 docs/pr67-normal-wal-reevaluation.md diff --git a/crates/cellule-runtime/src/node/bundle/binding.rs b/crates/cellule-runtime/src/node/bundle/binding.rs index a74fea2b..9dcd4e9a 100644 --- a/crates/cellule-runtime/src/node/bundle/binding.rs +++ b/crates/cellule-runtime/src/node/bundle/binding.rs @@ -30,6 +30,7 @@ impl NodeDirectory { predecessor: None, selected_through: 0, bindings: Vec::new(), + index: None, }; let prepared = self.upload_catalog(None, catalog, &[]).await?; self.select_catalog(observed, &prepared, now_ms).await @@ -50,8 +51,16 @@ impl NodeDirectory { .advertisement .bundle .ok_or(Error::Node("bundle lane is absent"))?; - let mut catalog = load_catalog(&self.layout, observed.advertisement.session, head).await?; let value = control.value(); + let shards = [index::shard( + authority.layout().application_id(), + value.cell.as_bytes(), + )] + .into_iter() + .collect(); + let mut catalog = + store::load_catalog_shards(&self.layout, observed.advertisement.session, head, &shards) + .await?; if value.state != ControlState::Serving || value.recovery.is_some() || value @@ -99,8 +108,10 @@ impl NodeDirectory { .control .bundle_binding .ok_or(Error::Node("provisional binding lacks pin"))? - } else if let Some(pin) = value.bundle_binding { - pin + } else if value.bundle_binding.is_some() { + // The provisional inventory CAS always precedes the original Cell + // pin CAS. A pinned Cell absent from its shard is not enrollment. + return Err(Error::Fenced); } else { let mut identity = value.encode()?; identity.extend_from_slice(b"cellule.bundle-binding.v1\0"); diff --git a/crates/cellule-runtime/src/node/bundle/closure.rs b/crates/cellule-runtime/src/node/bundle/closure.rs index 38b0e782..232fd778 100644 --- a/crates/cellule-runtime/src/node/bundle/closure.rs +++ b/crates/cellule-runtime/src/node/bundle/closure.rs @@ -13,55 +13,92 @@ impl NodeDirectory { &self, observed: &VersionedNodeAdvertisement, authority: &CellAuthority, - pin: BundleBindingRef, + proof: &BundleCoverageProof, + limits: cellule_ltx::Limits, + now_ms: i64, + ) -> Result { + self.checkpoint_bundle_cells(observed, &[(authority, proof)], limits, now_ms) + .await + } + + /// Checkpoints a bounded cohort of independently materialized roots with one + /// index upload and one shared node CAS. Each root must match its opaque + /// complete-capture proof; any missing/stale participant leaves the catalog + /// unchanged. Callers retain their admissions until this operation joins. + pub async fn checkpoint_bundle_cells( + &self, + observed: &VersionedNodeAdvertisement, + checkpoints: &[(&CellAuthority, &BundleCoverageProof)], limits: cellule_ltx::Limits, now_ms: i64, ) -> Result { + self.validate(&observed.advertisement, now_ms)?; + if checkpoints.is_empty() || checkpoints.len() > MAX_FRAMES { + return Err(Error::Capacity("bundle checkpoint count")); + } let head = observed .advertisement .bundle .ok_or(Error::Node("bundle lane is absent"))?; - let mut catalog = load_catalog(&self.layout, pin.session, head).await?; - if pin.epoch != catalog.epoch { - return Err(Error::Fenced); + let mut pins = std::collections::HashSet::new(); + let mut shards = std::collections::BTreeSet::new(); + for (_, proof) in checkpoints { + if proof.pin.session != observed.advertisement.session + || proof.pin.epoch != head.epoch + || !pins.insert(proof.pin.digest) + { + return Err(Error::Fenced); + } + shards.insert(index::shard( + proof.binding.application.as_bytes(), + proof.binding.control.cell.as_bytes(), + )); } - let binding = catalog.binding_mut(pin.digest)?; - if authority.layout().application_id() != binding.application.as_bytes() - || authority.layout().node_path(pin.session.as_bytes()) - != self.layout.node_path(pin.session.as_bytes()) - || authority.layout().immutable_cache_identity() - != self.layout.immutable_cache_identity() - { - return Err(Error::Fenced); + let mut catalog = + store::load_catalog_shards(&self.layout, observed.advertisement.session, head, &shards) + .await?; + let mut changed = false; + for (authority, proof) in checkpoints { + let pin = proof.binding(); + let binding = catalog.binding_mut(pin.digest)?; + if authority.layout().application_id() != binding.application.as_bytes() + || authority.layout().node_path(pin.session.as_bytes()) + != self.layout.node_path(pin.session.as_bytes()) + || authority.layout().immutable_cache_identity() + != self.layout.immutable_cache_identity() + { + return Err(Error::Fenced); + } + let current = authority + .load(binding.control.cell) + .await? + .ok_or(Error::Fenced)?; + let control = current.value(); + if control.bundle_binding != Some(pin) + || control.epoch != binding.control.epoch + || control.incarnation != binding.control.incarnation + || control.code != binding.control.code + || control.schema != binding.control.schema + { + return Err(Error::PendingPublication); + } + let root = control.ltx_root().ok_or(Error::PendingPublication)?; + let prefix = materialized_prefix(binding, proof, &root)?; + if prefix == 0 && Some(root) == binding.control.ltx_root() { + // Repeating an exact installed checkpoint is a no-op even if + // newer captures remain selected beyond this materialized base. + continue; + } + binding.control = control.clone(); + verify_base(&self.layout, binding, limits).await?; + binding.locators.drain(..prefix); + if binding.locators.is_empty() { + binding.first_commit = binding.selected_commit; + } + changed = true; } - let current = authority - .load(binding.control.cell) - .await? - .ok_or(Error::Fenced)?; - let control = current.value(); - if control.bundle_binding != Some(pin) - || control.epoch != binding.control.epoch - || control.incarnation != binding.control.incarnation - || control.code != binding.control.code - || control.schema != binding.control.schema - { - return Err(Error::PendingPublication); - } - let root = control.ltx_root().ok_or(Error::PendingPublication)?; - let frames = verify_binding(&self.layout, pin.session, head.epoch, binding, limits).await?; - let prefix = checkpoint_prefix(binding, &frames, &root)?; - if prefix == 0 && Some(root) == binding.control.ltx_root() { - return if binding.locators.is_empty() { - Ok(observed.clone()) - } else { - Err(Error::PendingPublication) - }; - } - binding.control = control.clone(); - verify_base(&self.layout, binding, limits).await?; - binding.locators.drain(..prefix); - if binding.locators.is_empty() { - binding.first_commit = binding.selected_commit; + if !changed { + return Ok(observed.clone()); } let prepared = self.upload_catalog(Some(head), catalog, &[]).await?; self.select_catalog(observed, &prepared, now_ms).await @@ -80,12 +117,19 @@ impl NodeDirectory { .advertisement .bundle .ok_or(Error::Node("bundle lane is absent"))?; - let mut catalog = load_catalog(&self.layout, pin.session, head).await?; + let scope = issued.scope(); + let shards = [index::shard( + scope.application.as_bytes(), + scope.cell.as_bytes(), + )] + .into_iter() + .collect(); + let mut catalog = + store::load_catalog_shards(&self.layout, pin.session, head, &shards).await?; if pin.epoch != catalog.epoch { return Err(Error::Fenced); } let binding = catalog.binding_mut(pin.digest)?; - let scope = issued.scope(); if issued.leader_session() != pin.session || issued.log_epoch() != pin.epoch || scope.application != binding.application @@ -117,15 +161,37 @@ impl NodeDirectory { &self, observed: &VersionedNodeAdvertisement, pin: BundleBindingRef, + issued: CellIssuedRange, now_ms: i64, ) -> Result { let head = observed .advertisement .bundle .ok_or(Error::Node("bundle lane is absent"))?; - let mut catalog = load_catalog(&self.layout, pin.session, head).await?; + let scope = issued.scope(); + let shards = [index::shard( + scope.application.as_bytes(), + scope.cell.as_bytes(), + )] + .into_iter() + .collect(); + let mut catalog = + store::load_catalog_shards(&self.layout, pin.session, head, &shards).await?; let binding = catalog.binding_mut(pin.digest)?; - if pin.epoch != head.epoch { + if pin.epoch != head.epoch + || issued.leader_session() != pin.session + || issued.log_epoch() != pin.epoch + || scope.application != binding.application + || scope.cell != binding.control.cell + || scope.incarnation != binding.control.incarnation + || scope.cell_epoch != binding.control.epoch + || binding.terminal + != Some(( + issued.last_node_sequence(), + issued.commit_sequence(), + issued.position(), + )) + { return Err(Error::Fenced); } if binding.phase == BindingPhase::Open @@ -143,3 +209,57 @@ impl NodeDirectory { self.select_catalog(observed, &prepared, now_ms).await } } + +/// The opaque proof authenticated a complete assigned capture endpoint before +/// selection. Matching its exact locator suffix against the fresh catalog +/// releases that prefix without re-reading data that materialization consumed. +/// A later checkpoint may have advanced the base; its retained locators must +/// still be the identical suffix of this proof. Root origin is verified by the +/// caller before the catalog CAS. An integer command watermark is insufficient. +fn materialized_prefix( + binding: &Binding, + proof: &BundleCoverageProof, + root: &cellule_ltx::RootRef, +) -> Result { + if binding.control.bundle_binding != Some(proof.pin) + || binding.application != proof.binding.application + || binding.control.cell != proof.binding.control.cell + || binding.control.incarnation != proof.binding.control.incarnation + || binding.control.epoch != proof.binding.control.epoch + || binding.control.code != proof.binding.control.code + || binding.control.schema != proof.binding.control.schema + || root.cell != *binding.control.cell.as_bytes() + || root.incarnation != *binding.control.incarnation.as_bytes() + { + return Err(Error::Fenced); + } + if root.commit_sequence != proof.commit_sequence() || root.position != proof.position() { + return Err(Error::PendingPublication); + } + let base = binding + .control + .ltx_root() + .ok_or(Error::PendingPublication)?; + if base.commit_sequence == root.commit_sequence && base.position == root.position { + return Ok(0); + } + if base.commit_sequence >= root.commit_sequence || base.position.txid >= root.position.txid { + return Err(Error::PendingPublication); + } + let first = binding.locators.first().ok_or(Error::PendingPublication)?; + let skip = proof + .binding + .locators + .iter() + .position(|locator| locator == first) + .ok_or(Error::Node( + "materialized proof does not continue the catalog base", + ))?; + let locators = &proof.binding.locators[skip..]; + if locators.is_empty() || binding.locators.get(..locators.len()) != Some(locators) { + return Err(Error::Node( + "materialized proof differs from selected catalog prefix", + )); + } + Ok(locators.len()) +} diff --git a/crates/cellule-runtime/src/node/bundle/codec.rs b/crates/cellule-runtime/src/node/bundle/codec.rs index 01ba2770..86248ac4 100644 --- a/crates/cellule-runtime/src/node/bundle/codec.rs +++ b/crates/cellule-runtime/src/node/bundle/codec.rs @@ -62,6 +62,7 @@ fn metadata(catalog: &Catalog, frames: usize) -> Result { Ok(e) } +#[cfg(test)] pub(super) fn encode( catalog: &mut Catalog, frames: &[cellule_ltx::VerifiedNodeFrame], @@ -99,7 +100,19 @@ pub(super) fn encode( Ok(Bytes::from(e.finish())) } +pub(super) fn encode_leaf(catalog: &Catalog) -> Result { + Ok(Bytes::from(metadata(catalog, 0)?.finish())) +} + pub(super) fn decode(body: &Bytes) -> Result { + decode_inner(body, true) +} + +pub(super) fn decode_leaf(body: &Bytes) -> Result { + decode_inner(body, false) +} + +fn decode_inner(body: &Bytes, complete_object: bool) -> Result { let mut d = BoundedDecoder::new(body, MAX_BUNDLE_BYTES as u32)?; if d.read_bytes()? != MAGIC { return Err(Error::Node("unsupported node bundle format")); @@ -178,7 +191,7 @@ pub(super) fn decode(body: &Bytes) -> Result { }); } let count = d.read_count()?; - if count > MAX_FRAMES { + if count > MAX_FRAMES || (!complete_object && count != 0) { return Err(Error::Capacity("bundle frame count")); } let mut local = Vec::with_capacity(count); @@ -200,6 +213,7 @@ pub(super) fn decode(body: &Bytes) -> Result { predecessor, selected_through, bindings, + index: None, }; catalog.validate()?; let locators: Vec<_> = catalog @@ -209,10 +223,11 @@ pub(super) fn decode(body: &Bytes) -> Result { .filter(|locator| locator.object.is_none()) .map(|locator| (locator.offset, locator.bytes, locator.frame_digest)) .collect(); - if locators.len() != local.len() - || local - .iter() - .any(|extent| locators.iter().filter(|locator| *locator == extent).count() != 1) + if complete_object + && (locators.len() != local.len() + || local + .iter() + .any(|extent| locators.iter().filter(|locator| *locator == extent).count() != 1)) { return Err(Error::Node( "bundle local extents differ from complete manifest", diff --git a/crates/cellule-runtime/src/node/bundle/index/codec.rs b/crates/cellule-runtime/src/node/bundle/index/codec.rs new file mode 100644 index 00000000..8218475e --- /dev/null +++ b/crates/cellule-runtime/src/node/bundle/index/codec.rs @@ -0,0 +1,179 @@ +//! Canonical fixed-size authenticated root header; leaf codecs remain CNB1. +use super::*; +use crate::codec::{BoundedDecoder, BoundedEncoder, read_fixed}; + +fn write_extent(e: &mut BoundedEncoder, extent: &Locator) -> Result<()> { + e.write_bool(extent.object.is_some())?; + if let Some(object) = extent.object { + e.write_bytes(object.as_bytes())?; + } + e.write_u64(extent.offset)?; + e.write_u64(extent.bytes)?; + e.write_bytes(extent.frame_digest.as_bytes())?; + Ok(()) +} +fn read_extent(d: &mut BoundedDecoder<'_>) -> Result { + let object = if d.read_bool()? { + Some(Digest::from_bytes(read_fixed( + d, + "bundle index object length", + )?)) + } else { + None + }; + Ok(Locator { + object, + offset: d.read_u64()?, + bytes: d.read_u64()?, + frame_digest: Digest::from_bytes(read_fixed(d, "bundle index digest length")?), + }) +} + +pub(super) fn encode(root: &Root) -> Result { + validate(root)?; + let mut e = BoundedEncoder::new((HEADER_BYTES - 12) as u32)?; + e.write_bytes(root.session.as_bytes())?; + e.write_u64(root.epoch)?; + e.write_bool(root.predecessor.is_some())?; + if let Some(previous) = root.predecessor { + e.write_bytes(previous.as_bytes())?; + } + e.write_u64(root.selected_through)?; + e.write_u64(root.object_bytes)?; + for shard in &root.shards { + e.write_bool(shard.is_some())?; + if let Some(shard) = shard { + write_extent(&mut e, &shard.extent)?; + e.write_count(shard.bindings)?; + } + } + e.write_count(root.frames.len())?; + for frame in &root.frames { + write_extent(&mut e, frame)?; + } + let payload = e.finish(); + let mut header = Vec::with_capacity(HEADER_BYTES); + header.extend_from_slice(MAGIC); + header.extend_from_slice(&(payload.len() as u32).to_be_bytes()); + header.extend_from_slice(&payload); + header.resize(HEADER_BYTES, 0); + Ok(Bytes::from(header)) +} + +pub(super) fn decode(header: &[u8]) -> Result { + if header.len() != HEADER_BYTES || header.get(..8) != Some(MAGIC.as_slice()) { + return Err(Error::Node("unsupported bundle index header")); + } + let length = u32::from_be_bytes( + header[8..12] + .try_into() + .map_err(|_| Error::Node("truncated bundle index header"))?, + ) as usize; + let end = 12_usize + .checked_add(length) + .filter(|end| *end <= HEADER_BYTES) + .ok_or(Error::Node("bundle index header length"))?; + if header[end..].iter().any(|byte| *byte != 0) { + return Err(Error::Node("noncanonical bundle index padding")); + } + let mut d = BoundedDecoder::new(&header[12..end], (HEADER_BYTES - 12) as u32)?; + let session = SessionId::from_bytes(read_fixed(&mut d, "bundle index session length")?); + let epoch = d.read_u64()?; + let predecessor = if d.read_bool()? { + Some(Digest::from_bytes(read_fixed( + &mut d, + "bundle index predecessor length", + )?)) + } else { + None + }; + let selected_through = d.read_u64()?; + let object_bytes = d.read_u64()?; + let mut shards = Vec::with_capacity(SHARDS); + for _ in 0..SHARDS { + shards.push(if d.read_bool()? { + Some(Shard { + extent: read_extent(&mut d)?, + bindings: d.read_count()?, + }) + } else { + None + }); + } + let count = d.read_count()?; + if count > MAX_FRAMES { + return Err(Error::Capacity("bundle index native frame count")); + } + let mut frames = Vec::with_capacity(count); + for _ in 0..count { + frames.push(read_extent(&mut d)?); + } + d.finish()?; + let root = Root { + session, + epoch, + predecessor, + selected_through, + object_bytes, + shards, + frames, + }; + validate(&root)?; + Ok(root) +} + +fn validate(root: &Root) -> Result<()> { + if root.epoch == 0 + || root.session.as_bytes().iter().all(|byte| *byte == 0) + || root.shards.len() != SHARDS + || root.frames.len() > MAX_FRAMES + || root.object_bytes < HEADER_BYTES as u64 + || root.object_bytes > MAX_BUNDLE_BYTES + { + return Err(Error::Node("invalid bundle index bounds")); + } + let mut bindings = 0_usize; + let mut local_end = HEADER_BYTES as u64; + for shard in root.shards.iter().flatten() { + bindings = bindings + .checked_add(shard.bindings) + .ok_or(Error::Capacity("bundle binding count"))?; + if shard.bindings == 0 || bindings > MAX_BINDINGS { + return Err(Error::Capacity("bundle binding count")); + } + validate_extent(&shard.extent)?; + if shard.extent.object.is_none() { + if shard.extent.offset != local_end { + return Err(Error::Node("noncanonical bundle index local shards")); + } + local_end += shard.extent.bytes; + } + } + for frame in &root.frames { + validate_extent(frame)?; + if frame.object.is_some() || frame.offset != local_end { + return Err(Error::Node("noncanonical bundle index native extents")); + } + local_end += frame.bytes; + } + if local_end != root.object_bytes { + return Err(Error::Node("bundle index object length differs")); + } + Ok(()) +} +fn validate_extent(extent: &Locator) -> Result<()> { + if extent.offset < HEADER_BYTES as u64 + || extent.bytes == 0 + || extent + .offset + .checked_add(extent.bytes) + .is_none_or(|end| end > MAX_BUNDLE_BYTES) + || extent.frame_digest.as_bytes().iter().all(|byte| *byte == 0) + || extent + .object + .is_some_and(|digest| digest.as_bytes().iter().all(|byte| *byte == 0)) + { + return Err(Error::Node("invalid bundle index extent")); + } + Ok(()) +} diff --git a/crates/cellule-runtime/src/node/bundle/index/io.rs b/crates/cellule-runtime/src/node/bundle/index/io.rs new file mode 100644 index 00000000..3a74dc32 --- /dev/null +++ b/crates/cellule-runtime/src/node/bundle/index/io.rs @@ -0,0 +1,162 @@ +//! Bounded origin point lookups and streaming maintenance inventory checks. +use super::*; + +async fn load_root( + layout: &cellule_ltx::CellStorageLayout, + session: SessionId, + head: NodeBundleHead, +) -> Result> { + let path = + layout.node_coverage_bundle_path(session.as_bytes(), head.epoch, head.digest.as_bytes()); + let header = layout + .store() + .range_get(&path, 0..HEADER_BYTES as u64) + .await?; + if header.get(..8) != Some(MAGIC.as_slice()) { + return Ok(None); + } + if *blake3::hash(&header).as_bytes() != *head.digest.as_bytes() { + return Err(Error::Node("bundle index header digest differs")); + } + let mut root = codec::decode(&header)?; + if root.session != session + || root.epoch != head.epoch + || root.selected_through != head.selected_through + { + return Err(Error::Node("bundle index scope differs")); + } + for shard in root.shards.iter_mut().flatten() { + if shard.extent.object.is_none() { + shard.extent.object = Some(head.digest); + } + } + Ok(Some(root)) +} + +async fn load_rows( + layout: &cellule_ltx::CellStorageLayout, + root: &Root, + shard: &Shard, + id: u8, +) -> Result> { + let extent = &shard.extent; + let object = extent + .object + .ok_or(Error::Node("unresolved bundle catalog shard"))?; + let bytes = layout + .store() + .range_get( + &layout.node_coverage_bundle_path( + root.session.as_bytes(), + root.epoch, + object.as_bytes(), + ), + extent.offset..extent.offset + extent.bytes, + ) + .await?; + let mut leaf = decode_leaf(bytes, shard, id, root)?; + for locator in leaf + .bindings + .iter_mut() + .flat_map(|binding| &mut binding.locators) + { + if locator.object.is_none() { + locator.object = Some(object); + } + } + Ok(leaf.bindings) +} + +pub(in crate::node::bundle) async fn load( + layout: &cellule_ltx::CellStorageLayout, + session: SessionId, + head: NodeBundleHead, + wanted: Option<&BTreeSet>, +) -> Result { + let Some(root) = load_root(layout, session, head).await? else { + return super::super::store::load_legacy_catalog(layout, session, head).await; + }; + // An immutable shard can be individually bounded while a selected cohort + // spans many large historical shards. Charge its aggregate encoded metadata + // before any leaf I/O or decoding; a point lookup charges only its shard. + let mut selected_bytes = 0_u64; + for (id, shard) in root.shards.iter().enumerate() { + if wanted.is_some_and(|wanted| !wanted.contains(&(id as u8))) { + continue; + } + selected_bytes = selected_bytes + .checked_add(shard.as_ref().map_or(0, |shard| shard.extent.bytes)) + .filter(|bytes| *bytes <= MAX_BUNDLE_BYTES) + .ok_or(Error::Capacity("bundle selected shard bytes"))?; + } + let mut loaded = BTreeMap::new(); + let mut bindings = Vec::new(); + for id in 0..SHARDS { + let id = id as u8; + if wanted.is_some_and(|wanted| !wanted.contains(&id)) { + continue; + } + let rows = match &root.shards[usize::from(id)] { + Some(shard) => load_rows(layout, &root, shard, id).await?, + None => Vec::new(), + }; + bindings.extend(rows.iter().cloned()); + loaded.insert(id, rows); + } + bindings.sort_unstable_by_key(|binding| { + binding + .control + .bundle_binding + .map(|pin| *pin.digest.as_bytes()) + }); + let catalog = Catalog { + session, + epoch: head.epoch, + predecessor: root.predecessor, + selected_through: head.selected_through, + bindings, + index: Some(LoadedIndex { root, loaded }), + }; + catalog.validate()?; + Ok(catalog) +} + +/// A clean maintenance result covers every indexed binding. Only one bounded +/// shard's decoded rows are retained at a time; the global duplicate-pin set is +/// bounded by MAX_BINDINGS. No missing shard is interpreted as an empty shard. +pub(in crate::node::bundle) async fn ensure_drained( + layout: &cellule_ltx::CellStorageLayout, + session: SessionId, + head: NodeBundleHead, +) -> Result<()> { + let Some(root) = load_root(layout, session, head).await? else { + let catalog = super::super::store::load_legacy_catalog(layout, session, head).await?; + return check_closed(&catalog.bindings); + }; + let mut pins = std::collections::HashSet::new(); + for (id, shard) in root.shards.iter().enumerate() { + let Some(shard) = shard else { continue }; + let rows = load_rows(layout, &root, shard, id as u8).await?; + for binding in &rows { + let pin = binding + .control + .bundle_binding + .ok_or(Error::Node("bundle catalog lacks Cell pin"))?; + if !pins.insert(pin.digest) { + return Err(Error::Node("bundle inventory repeats a Cell pin")); + } + } + check_closed(&rows)?; + } + Ok(()) +} + +fn check_closed(bindings: &[Binding]) -> Result<()> { + if bindings + .iter() + .any(|binding| binding.phase != BindingPhase::Closed || !binding.locators.is_empty()) + { + return Err(Error::PendingPublication); + } + Ok(()) +} diff --git a/crates/cellule-runtime/src/node/bundle/index/mod.rs b/crates/cellule-runtime/src/node/bundle/index/mod.rs new file mode 100644 index 00000000..648d05cf --- /dev/null +++ b/crates/cellule-runtime/src/node/bundle/index/mod.rs @@ -0,0 +1,300 @@ +//! Copy-on-write authenticated catalog shards inside the selected bundle object. +//! +//! The node head authenticates a fixed header. That header authenticates each +//! shard extent and every new native extent. Unchanged shards keep their exact +//! object/range/digest, so neither publication nor a point lookup walks a chain. +use super::*; +use std::collections::{BTreeMap, BTreeSet}; + +mod codec; +mod io; +#[cfg(test)] +mod tests; +pub(super) use io::{ensure_drained, load}; + +pub(super) const HEADER_BYTES: usize = 32 << 10; +const SHARDS: usize = 256; +const MAGIC: &[u8; 8] = b"\0\0\0\x04CNB2"; + +#[derive(Clone, Debug, PartialEq, Eq)] +struct Shard { + extent: Locator, + bindings: usize, +} + +#[derive(Clone, Debug, PartialEq, Eq)] +struct Root { + session: SessionId, + epoch: u64, + predecessor: Option, + selected_through: u64, + object_bytes: u64, + shards: Vec>, + frames: Vec, +} + +#[derive(Clone, Debug, PartialEq, Eq)] +pub(super) struct LoadedIndex { + root: Root, + loaded: BTreeMap>, +} + +/// The shard key includes the application, so equal Cell IDs in applications +/// never alias. Epoch/incarnation remain inside the authenticated binding rows. +pub(super) fn shard(application: &[u8; 16], cell: &[u8; 32]) -> u8 { + let mut hash = blake3::Hasher::new(); + hash.update(b"cellule.bundle-catalog-shard.v2\0"); + hash.update(application); + hash.update(cell); + hash.finalize().as_bytes()[0] +} + +fn binding_shard(binding: &Binding) -> u8 { + shard( + binding.application.as_bytes(), + binding.control.cell.as_bytes(), + ) +} + +fn grouped(catalog: &Catalog) -> BTreeMap> { + let mut groups = BTreeMap::>::new(); + for binding in &catalog.bindings { + groups + .entry(binding_shard(binding)) + .or_default() + .push(binding.clone()); + } + groups +} + +fn leaf(catalog: &Catalog, bindings: Vec) -> Catalog { + Catalog { + session: catalog.session, + epoch: catalog.epoch, + predecessor: None, + selected_through: catalog.selected_through, + bindings, + index: None, + } +} + +pub(super) fn encode( + catalog: &mut Catalog, + frames: &[cellule_ltx::VerifiedNodeFrame], +) -> Result<(Bytes, Digest)> { + catalog.validate()?; + if frames.len() > MAX_FRAMES { + return Err(Error::Capacity("bundle frame count")); + } + let groups = grouped(catalog); + let mut shards = match &catalog.index { + Some(index) => { + if index.root.session != catalog.session || index.root.epoch != catalog.epoch { + return Err(Error::Fenced); + } + // A partial catalog may modify only a shard it authenticated first. + if groups.keys().any(|id| !index.loaded.contains_key(id)) { + return Err(Error::Node("bundle modifies an unloaded catalog shard")); + } + index.root.shards.clone() + } + None => vec![None; SHARDS], + }; + let changed: BTreeSet = match &catalog.index { + Some(index) => index + .loaded + .iter() + .filter_map(|(id, original)| { + (groups.get(id).map(Vec::as_slice).unwrap_or(&[]) != original.as_slice()) + .then_some(*id) + }) + .collect(), + None => groups.keys().copied().collect(), + }; + // Native extents follow the changed shards; changing numeric offsets never + // changes the leaf encoding length. Compute sizes before filling offsets. + let mut offset = HEADER_BYTES as u64; + for id in &changed { + let rows = groups.get(id).cloned().unwrap_or_default(); + if rows.is_empty() { + shards[usize::from(*id)] = None; + } else { + let bytes = super::codec::encode_leaf(&leaf(catalog, rows))?.len() as u64; + shards[usize::from(*id)] = Some(Shard { + extent: Locator { + object: None, + offset, + bytes, + frame_digest: Digest::from_bytes([0; 32]), + }, + bindings: groups[id].len(), + }); + offset = offset + .checked_add(bytes) + .ok_or(Error::Capacity("bundle shard bytes"))?; + } + } + let mut native = Vec::with_capacity(frames.len()); + for frame in frames { + let digest = Digest::from_bytes(frame.digest()); + let mut matches = 0; + for binding in &mut catalog.bindings { + let id = binding_shard(binding); + for locator in &mut binding.locators { + if locator.object.is_none() && locator.frame_digest == digest { + if !changed.contains(&id) { + return Err(Error::Node("native frame belongs to an unchanged shard")); + } + locator.offset = offset; + locator.bytes = frame.encoded().len() as u64; + matches += 1; + } + } + } + if matches != 1 { + return Err(Error::Node("bundle frame locator is not unique")); + } + native.push(Locator { + object: None, + offset, + bytes: frame.encoded().len() as u64, + frame_digest: digest, + }); + offset = offset + .checked_add(frame.encoded().len() as u64) + .ok_or(Error::Capacity("bundle bytes"))?; + } + if offset > MAX_BUNDLE_BYTES { + return Err(Error::Capacity("bundle bytes")); + } + let groups = grouped(catalog); + let mut bodies = Vec::new(); + for id in &changed { + if let Some(reference) = &mut shards[usize::from(*id)] { + let body = super::codec::encode_leaf(&leaf(catalog, groups[id].clone()))?; + if body.len() as u64 != reference.extent.bytes { + return Err(Error::Node("bundle shard size changed during encoding")); + } + reference.extent.frame_digest = Digest::from_bytes(*blake3::hash(&body).as_bytes()); + bodies.push(body); + } + } + let root = Root { + session: catalog.session, + epoch: catalog.epoch, + predecessor: catalog.predecessor, + selected_through: catalog.selected_through, + object_bytes: offset, + shards, + frames: native, + }; + let header = codec::encode(&root)?; + let digest = Digest::from_bytes(*blake3::hash(&header).as_bytes()); + let mut body = Vec::with_capacity(offset as usize); + body.extend_from_slice(&header); + for leaf in bodies { + body.extend_from_slice(&leaf); + } + for frame in frames { + body.extend_from_slice(frame.encoded()); + } + let body = Bytes::from(body); + verify_local(&body, &root)?; + Ok((body, digest)) +} + +/// Self-verification covers all bytes of a newly uploaded object. References +/// to old immutable shards remain authenticated by the predecessor selection. +fn verify_local(body: &Bytes, expected: &Root) -> Result<()> { + let root = codec::decode(&body[..HEADER_BYTES])?; + if &root != expected || body.len() as u64 != root.object_bytes { + return Err(Error::Node("bundle index self-verification differs")); + } + let mut offset = HEADER_BYTES as u64; + let mut locators = Vec::new(); + for (id, shard) in root.shards.iter().enumerate() { + let Some(shard) = shard else { continue }; + if shard.extent.object.is_some() { + continue; + } + if shard.extent.offset != offset { + return Err(Error::Node("bundle local shard extents are not canonical")); + } + let bytes = extent_bytes(body, &shard.extent)?; + let leaf = decode_leaf(bytes, shard, id as u8, &root)?; + locators.extend( + leaf.bindings + .into_iter() + .flat_map(|binding| binding.locators) + .filter(|locator| locator.object.is_none()), + ); + offset += shard.extent.bytes; + } + if locators.len() != root.frames.len() { + return Err(Error::Node("bundle local native extent count differs")); + } + for frame in &root.frames { + if frame.offset != offset + || locators.iter().filter(|locator| *locator == frame).count() != 1 + { + return Err(Error::Node("bundle local native extents differ")); + } + extent_bytes(body, frame)?; + offset += frame.bytes; + } + if offset != root.object_bytes { + return Err(Error::Node("bundle has unauthenticated trailing bytes")); + } + Ok(()) +} + +fn extent_bytes(body: &Bytes, locator: &Locator) -> Result { + let end = locator + .offset + .checked_add(locator.bytes) + .ok_or(Error::Node("bundle extent overflow"))?; + let bytes = body + .get(locator.offset as usize..end as usize) + .ok_or(Error::Node("bundle extent is truncated"))?; + if *blake3::hash(bytes).as_bytes() != *locator.frame_digest.as_bytes() { + return Err(Error::Node("bundle indexed extent digest differs")); + } + Ok(Bytes::copy_from_slice(bytes)) +} + +fn decode_leaf(body: Bytes, shard: &Shard, id: u8, root: &Root) -> Result { + if body.len() as u64 != shard.extent.bytes + || *blake3::hash(&body).as_bytes() != *shard.extent.frame_digest.as_bytes() + { + return Err(Error::Node("bundle catalog shard digest differs")); + } + let catalog = super::codec::decode_leaf(&body)?; + if catalog.session != root.session + || catalog.epoch != root.epoch + || catalog.selected_through > root.selected_through + || catalog.predecessor.is_some() + || catalog.bindings.len() != shard.bindings + || catalog + .bindings + .iter() + .any(|binding| binding_shard(binding) != id) + { + return Err(Error::Node("bundle catalog shard scope differs")); + } + Ok(catalog) +} + +pub(super) fn body_digest(body: &Bytes) -> Result { + if body.get(..8) == Some(MAGIC.as_slice()) { + let header = body + .get(..HEADER_BYTES) + .ok_or(Error::Node("truncated bundle index header"))?; + let root = codec::decode(header)?; + if root.object_bytes != body.len() as u64 { + return Err(Error::Node("bundle proposal object length differs")); + } + Ok(Digest::from_bytes(*blake3::hash(header).as_bytes())) + } else { + Ok(Digest::from_bytes(*blake3::hash(body).as_bytes())) + } +} diff --git a/crates/cellule-runtime/src/node/bundle/index/tests.rs b/crates/cellule-runtime/src/node/bundle/index/tests.rs new file mode 100644 index 00000000..fb058adf --- /dev/null +++ b/crates/cellule-runtime/src/node/bundle/index/tests.rs @@ -0,0 +1,70 @@ +use super::*; +use cellule_store::test_support::CountingObjectStore; +use object_store::{memory::InMemory, path::Path}; +use std::sync::Arc; + +#[tokio::test] +async fn selected_shard_budget_is_checked_before_loading_any_leaf() { + let counted = Arc::new(CountingObjectStore::new(Arc::new(InMemory::new()))); + let layout = cellule_ltx::CellStorageLayout::new( + cellule_store::Store::new(counted.clone()), + Path::from("index-admission"), + [9; 16], + ); + let session = SessionId::from_bytes([1; 16]); + let mut shards = vec![None; SHARDS]; + // Both extents are individually bounded. Neither missing leaf should be + // requested when their selected aggregate already exceeds the budget. + for id in [0, 1] { + shards[id] = Some(Shard { + extent: Locator { + object: Some(Digest::from_bytes([id as u8 + 2; 32])), + offset: HEADER_BYTES as u64, + bytes: 3 << 20, + frame_digest: Digest::from_bytes([id as u8 + 4; 32]), + }, + bindings: 1, + }); + } + let root = Root { + session, + epoch: 2, + predecessor: None, + selected_through: 0, + object_bytes: HEADER_BYTES as u64, + shards, + frames: Vec::new(), + }; + let header = codec::encode(&root).unwrap(); + let head = NodeBundleHead { + epoch: 2, + digest: Digest::from_bytes(*blake3::hash(&header).as_bytes()), + selected_through: 0, + }; + layout + .store() + .put_exact( + &layout.node_coverage_bundle_path(session.as_bytes(), 2, head.digest.as_bytes()), + header, + ) + .await + .unwrap(); + counted.reset(); + assert!(matches!( + load(&layout, session, head, Some(&[0, 1].into_iter().collect())).await, + Err(Error::Capacity("bundle selected shard bytes")) + )); + assert_eq!( + counted.requests().len(), + 1, + "only the authenticated header is fetched" + ); + // A point lookup charges only its chosen shard. Other large shards are + // not a reason to reject it; this request reaches the missing origin leaf. + counted.reset(); + assert!(!matches!( + load(&layout, session, head, Some(&[0].into_iter().collect())).await, + Err(Error::Capacity("bundle selected shard bytes")) + )); + assert_eq!(counted.requests().len(), 2); +} diff --git a/crates/cellule-runtime/src/node/bundle/mod.rs b/crates/cellule-runtime/src/node/bundle/mod.rs index c03638a2..b3fa0e1a 100644 --- a/crates/cellule-runtime/src/node/bundle/mod.rs +++ b/crates/cellule-runtime/src/node/bundle/mod.rs @@ -38,9 +38,13 @@ use bytes::Bytes; mod binding; mod closure; mod codec; +mod index; mod proof; mod selection; -use proof::{checkpoint_prefix, verify_base, verify_binding}; +#[cfg(test)] +use proof::checkpoint_prefix; +use proof::{verify_base, verify_binding}; +#[cfg(test)] use store::load_catalog; pub(crate) mod store; #[cfg(test)] @@ -67,7 +71,7 @@ impl NodeBundleHead { pub const fn epoch(&self) -> u64 { self.epoch } - /// Digest of the complete immutable catalog and range. + /// Digest authenticating the immutable catalog index and exact range. pub const fn digest(&self) -> Digest { self.digest } @@ -121,6 +125,9 @@ struct Catalog { predecessor: Option, selected_through: u64, bindings: Vec, + // Authenticated unchanged shards survive a partial load. Loaded rows are + // snapshots for copy-on-write comparison, never a second authority. + index: Option, } /// Uploaded exact proposal. It cannot release an ACK or prune a capture. @@ -189,7 +196,10 @@ impl Catalog { .ok_or(Error::Node("bundle catalog lacks base"))?; if pin.session != self.session || pin.epoch != self.epoch - || binding.first_commit == 0 + // A freshly published runtime bootstrap has no logical command + // or assigned frame yet. Only that empty binding may use zero. + || (binding.first_commit == 0 + && (binding.selected_commit != 0 || binding.selected_sequence != 0)) || binding.first_commit > binding.selected_commit || !scopes.insert(( binding.application, diff --git a/crates/cellule-runtime/src/node/bundle/proof.rs b/crates/cellule-runtime/src/node/bundle/proof.rs index 58b2909b..8bbfd449 100644 --- a/crates/cellule-runtime/src/node/bundle/proof.rs +++ b/crates/cellule-runtime/src/node/bundle/proof.rs @@ -41,7 +41,13 @@ impl NodeDirectory { return Err(Error::Fenced); } let head = head.ok_or(Error::PendingPublication)?; - let mut catalog = load_catalog(&self.layout, session, head).await?; + let shards = [index::shard( + authority.layout().application_id(), + value.cell.as_bytes(), + )] + .into_iter() + .collect(); + let mut catalog = store::load_catalog_shards(&self.layout, session, head, &shards).await?; let binding = catalog.binding_mut(pin.digest)?.clone(); if binding.phase == BindingPhase::Provisional { return Err(Error::PendingPublication); diff --git a/crates/cellule-runtime/src/node/bundle/selection.rs b/crates/cellule-runtime/src/node/bundle/selection.rs index 194161a1..1a4a29f3 100644 --- a/crates/cellule-runtime/src/node/bundle/selection.rs +++ b/crates/cellule-runtime/src/node/bundle/selection.rs @@ -18,10 +18,19 @@ impl NodeDirectory { .advertisement .bundle .ok_or(Error::Node("bundle lane is absent"))?; - let mut catalog = load_catalog(&self.layout, observed.advertisement.session, head).await?; if frames.is_empty() || frames.len() > MAX_FRAMES { return Err(Error::Capacity("bundle frame count")); } + let shards = frames + .iter() + .map(|frame| { + let scope = frame.scope(); + index::shard(&scope.application, &scope.cell) + }) + .collect(); + let mut catalog = + store::load_catalog_shards(&self.layout, observed.advertisement.session, head, &shards) + .await?; let mut consumed = 0_usize; for assignment in assignments { let count = usize::try_from( @@ -112,7 +121,24 @@ impl NodeDirectory { now_ms: i64, ) -> Result<(VersionedNodeAdvertisement, Vec)> { lease.check()?; - let catalog = load_catalog(&self.layout, prepared.catalog.session, prepared.head).await?; + let shards = prepared + .catalog + .bindings + .iter() + .map(|binding| { + index::shard( + binding.application.as_bytes(), + binding.control.cell.as_bytes(), + ) + }) + .collect(); + let catalog = store::load_catalog_shards( + &self.layout, + prepared.catalog.session, + prepared.head, + &shards, + ) + .await?; let mut proofs = Vec::new(); for binding in catalog.bindings { if !binding diff --git a/crates/cellule-runtime/src/node/bundle/store.rs b/crates/cellule-runtime/src/node/bundle/store.rs index a6903882..b0facd29 100644 --- a/crates/cellule-runtime/src/node/bundle/store.rs +++ b/crates/cellule-runtime/src/node/bundle/store.rs @@ -16,14 +16,10 @@ impl NodeDirectory { )); } catalog.predecessor = original.map(|head| head.digest); - let body = codec::encode(&mut catalog, frames)?; - // Decode the very bytes sent to the provider before selecting them. - if codec::decode(&body)? != catalog { - return Err(Error::Node("bundle manifest self-verification differs")); - } + let (body, digest) = index::encode(&mut catalog, frames)?; let head = NodeBundleHead { epoch: catalog.epoch, - digest: Digest::from_bytes(*blake3::hash(&body).as_bytes()), + digest, selected_through: catalog.selected_through, }; self.layout @@ -52,7 +48,7 @@ impl NodeDirectory { now_ms: i64, ) -> Result { let mut base = observed.clone(); - if *blake3::hash(&prepared.body).as_bytes() != *prepared.head.digest.as_bytes() { + if index::body_digest(&prepared.body)? != prepared.head.digest { return Err(Error::Node("bundle proposal digest differs")); } let mut last_conflict = None; @@ -110,10 +106,28 @@ impl NodeDirectory { } } +#[cfg(test)] pub(super) async fn load_catalog( layout: &cellule_ltx::CellStorageLayout, session: SessionId, head: NodeBundleHead, +) -> Result { + index::load(layout, session, head, None).await +} + +pub(super) async fn load_catalog_shards( + layout: &cellule_ltx::CellStorageLayout, + session: SessionId, + head: NodeBundleHead, + shards: &std::collections::BTreeSet, +) -> Result { + index::load(layout, session, head, Some(shards)).await +} + +pub(super) async fn load_legacy_catalog( + layout: &cellule_ltx::CellStorageLayout, + session: SessionId, + head: NodeBundleHead, ) -> Result { let (body, _) = layout .store() @@ -172,7 +186,13 @@ pub(crate) async fn ensure_enrollment( return Err(Error::Fenced); } let head = node.bundle.ok_or(Error::PendingPublication)?; - let mut catalog = load_catalog(layout, pin.session, head).await?; + let shards = [index::shard( + layout.application_id(), + control.cell.as_bytes(), + )] + .into_iter() + .collect(); + let mut catalog = load_catalog_shards(layout, pin.session, head, &shards).await?; let binding = catalog.binding_mut(pin.digest)?; if head.epoch != pin.epoch || binding.phase != BindingPhase::Provisional @@ -216,7 +236,13 @@ pub(crate) async fn ensure_departure( return Err(Error::Fenced); } let head = head.ok_or(Error::PendingPublication)?; - let mut catalog = load_catalog(layout, pin.session, head).await?; + let shards = [index::shard( + layout.application_id(), + control.cell.as_bytes(), + )] + .into_iter() + .collect(); + let mut catalog = load_catalog_shards(layout, pin.session, head, &shards).await?; if pin.epoch != catalog.epoch { return Err(Error::Fenced); } @@ -248,13 +274,5 @@ pub(crate) async fn ensure_session_drained( let Some(head) = head else { return Ok(()); }; - let catalog = load_catalog(layout, session, head).await?; - if catalog - .bindings - .iter() - .any(|binding| binding.phase != BindingPhase::Closed || !binding.locators.is_empty()) - { - return Err(Error::PendingPublication); - } - Ok(()) + index::ensure_drained(layout, session, head).await } diff --git a/crates/cellule-runtime/src/node/bundle/tests/index/bootstrap.rs b/crates/cellule-runtime/src/node/bundle/tests/index/bootstrap.rs new file mode 100644 index 00000000..187fc723 --- /dev/null +++ b/crates/cellule-runtime/src/node/bundle/tests/index/bootstrap.rs @@ -0,0 +1,87 @@ +use super::*; + +#[tokio::test] +async fn zero_command_bootstrap_enrolls_and_materializes_the_first_command() { + let mut f = Fixture::new().await; + let mut cell = f.unbound_cell_at_commit(4, [9; 16], 0).await; + let initial = cell.control.value().ltx_root().unwrap(); + assert_eq!(initial.commit_sequence, 0); + let (node, control) = f + .directory + .bind_bundle_cell(&f.node, &cell.authority, &cell.control, NOW) + .await + .unwrap(); + f.node = node; + cell.control = control; + let (_, frames, assigned) = f.append(&mut cell, 1); + let proposal = f + .directory + .prepare_node_bundle(&f.node, &frames, &[assigned], NOW) + .await + .unwrap(); + let (node, mut proofs) = f + .directory + .select_node_bundle(&f.node, &proposal, &f.lease, Limits::default(), NOW) + .await + .unwrap(); + let proof = proofs.pop().unwrap(); + assert_eq!(proof.base().unwrap(), initial); + assert_eq!(proof.commit_sequence(), 1); + let root = f.publisher(&cell).materialize_bundle(&proof).await.unwrap(); + assert_eq!(root.commit_sequence, 1); + assert_eq!(root.position, proof.position()); + f.directory + .checkpoint_bundle_cell(&node, &cell.authority, &proof, Limits::default(), NOW) + .await + .unwrap(); + let file = f.scratch.path().join("bootstrap-restored.sqlite"); + cell.replica + .open_root(&root) + .await + .unwrap() + .restore(&file) + .await + .unwrap(); + let db = rusqlite::Connection::open(file).unwrap(); + let result: String = db + .query_row( + "SELECT result FROM outcomes WHERE request='request-1'", + [], + |row| row.get(0), + ) + .unwrap(); + assert_eq!(result, "result-1"); +} + +#[tokio::test] +async fn quiet_zero_command_bootstrap_can_drain_without_inventing_an_issued_command() { + let mut f = Fixture::new().await; + let mut cell = f.unbound_cell_at_commit(4, [9; 16], 0).await; + let (node, control) = f + .directory + .bind_bundle_cell(&f.node, &cell.authority, &cell.control, NOW) + .await + .unwrap(); + cell.control = control; + let pin = cell.control.value().bundle_binding.unwrap(); + let issued = f + .gate + .close_cell_issuance( + Fixture::scope(&cell), + cell.control.value().ltx_root().unwrap(), + ) + .unwrap(); + assert_eq!(issued.commit_sequence(), 0); + assert_eq!(issued.last_node_sequence(), 0); + let closing = f + .directory + .begin_bundle_close(&node, pin, issued, NOW) + .await + .unwrap(); + let closed = f + .directory + .finish_bundle_close(&closing, pin, issued, NOW) + .await + .unwrap(); + f.directory.withdraw(&closed, NOW).await.unwrap(); +} diff --git a/crates/cellule-runtime/src/node/bundle/tests/index/checkpoint.rs b/crates/cellule-runtime/src/node/bundle/tests/index/checkpoint.rs new file mode 100644 index 00000000..df74c44e --- /dev/null +++ b/crates/cellule-runtime/src/node/bundle/tests/index/checkpoint.rs @@ -0,0 +1,133 @@ +use super::*; + +#[tokio::test] +async fn checkpoint_uses_the_exact_materialized_proof_without_scanning_siblings_or_old_frames() { + let mut f = Fixture::new().await; + let mut cell = f.cell(4).await; + inventory(&mut f, &cell, 1_000).await; + let (_, frames, assigned) = f.append(&mut cell, 2); + let proposal = f + .directory + .prepare_node_bundle(&f.node, &frames, &[assigned], NOW) + .await + .unwrap(); + let (node, mut proofs) = f + .directory + .select_node_bundle(&f.node, &proposal, &f.lease, Limits::default(), NOW) + .await + .unwrap(); + let old = proofs.pop().unwrap(); + f.node = node; + let mut publisher = f.publisher(&cell); + let materialized = publisher.materialize_bundle(&old).await.unwrap(); + let (_, frames, assigned) = f.append(&mut cell, 3); + let proposal = f + .directory + .prepare_node_bundle(&f.node, &frames, &[assigned], NOW) + .await + .unwrap(); + let (node, _) = f + .directory + .select_node_bundle(&f.node, &proposal, &f.lease, Limits::default(), NOW) + .await + .unwrap(); + f.node = node; + f.count.reset(); + let checkpoint = f + .directory + .checkpoint_bundle_cell(&f.node, &cell.authority, &old, Limits::default(), NOW) + .await + .unwrap(); + let reads: Vec<_> = f + .count + .requests() + .into_iter() + .filter(|read| read.location.ends_with(".cnb")) + .collect(); + assert_eq!( + reads.len(), + 2, + "checkpoint should read only its authenticated header and chosen shard" + ); + assert_eq!(f.count.put_requests(), 2); + f.node = checkpoint; + // A lost caller reply may retry the same exact checkpoint against a newer + // node observation. The already installed base leaves the hot suffix intact. + f.count.reset(); + let repeated = f + .directory + .checkpoint_bundle_cell(&f.node, &cell.authority, &old, Limits::default(), NOW) + .await + .unwrap(); + assert_eq!( + repeated.advertisement().bundle_head(), + f.node.advertisement().bundle_head() + ); + assert_eq!(f.count.put_requests(), 0); + let current = cell + .authority + .load(cell.control.value().cell) + .await + .unwrap() + .unwrap(); + let suffix = f + .directory + .load_bundle_coverage(&cell.authority, ¤t, Limits::default()) + .await + .unwrap(); + assert_eq!(suffix.base().unwrap(), materialized); + assert_eq!(suffix.commit_sequence(), 3); + assert_eq!(suffix.locator_count(), frames.len()); +} + +#[tokio::test] +async fn a_stale_proof_cannot_checkpoint_a_newer_materialized_endpoint() { + let mut f = Fixture::new().await; + let mut cell = f.cell(4).await; + let (_, frames, assigned) = f.append(&mut cell, 2); + let proposal = f + .directory + .prepare_node_bundle(&f.node, &frames, &[assigned], NOW) + .await + .unwrap(); + let (node, mut proofs) = f + .directory + .select_node_bundle(&f.node, &proposal, &f.lease, Limits::default(), NOW) + .await + .unwrap(); + let old = proofs.pop().unwrap(); + f.node = node; + let (_, frames, assigned) = f.append(&mut cell, 3); + let proposal = f + .directory + .prepare_node_bundle(&f.node, &frames, &[assigned], NOW) + .await + .unwrap(); + let (node, mut proofs) = f + .directory + .select_node_bundle(&f.node, &proposal, &f.lease, Limits::default(), NOW) + .await + .unwrap(); + let latest = proofs.pop().unwrap(); + f.node = node; + f.publisher(&cell) + .materialize_bundle(&latest) + .await + .unwrap(); + f.count.reset(); + assert!(matches!( + f.directory + .checkpoint_bundle_cell(&f.node, &cell.authority, &old, Limits::default(), NOW) + .await, + Err(Error::PendingPublication) + )); + assert_eq!( + f.count.put_requests(), + 0, + "a stale endpoint must not change the catalog" + ); + f.directory + .checkpoint_bundle_cell(&f.node, &cell.authority, &latest, Limits::default(), NOW) + .await + .unwrap(); +} diff --git a/crates/cellule-runtime/src/node/bundle/tests/index/cohort.rs b/crates/cellule-runtime/src/node/bundle/tests/index/cohort.rs new file mode 100644 index 00000000..00c2f42c --- /dev/null +++ b/crates/cellule-runtime/src/node/bundle/tests/index/cohort.rs @@ -0,0 +1,176 @@ +use super::*; + +#[tokio::test] +async fn checkpoint_cohort_uses_two_puts_and_preserves_a_hot_cell_suffix() { + let mut f = Fixture::new().await; + let mut cells = Vec::new(); + for number in 4..(4 + MAX_FRAMES as u8) { + cells.push(f.cell(number).await); + } + let mut frames = Vec::new(); + let mut assigned = Vec::new(); + for cell in &mut cells { + let (_, capture, range) = f.append(cell, 2); + frames.extend(capture); + assigned.push(range); + } + let proposal = f + .directory + .prepare_node_bundle(&f.node, &frames, &assigned, NOW) + .await + .unwrap(); + let (node, proofs) = f + .directory + .select_node_bundle(&f.node, &proposal, &f.lease, Limits::default(), NOW) + .await + .unwrap(); + f.node = node; + assert_eq!(proofs.len(), MAX_FRAMES); + f.count.reset(); + for cell in &cells { + let proof = proofs + .iter() + .find(|proof| proof.binding() == cell.control.value().bundle_binding.unwrap()) + .unwrap(); + f.publisher(cell).materialize_bundle(proof).await.unwrap(); + } + eprintln!( + "materialization cohort: cells={} puts={}", + cells.len(), + f.count.put_requests() + ); + assert_eq!( + f.count.put_requests(), + MAX_FRAMES * 4, + "a singleton materialized suffix should use the canonical native pack" + ); + let (_, frames, assigned) = f.append(&mut cells[0], 3); + let proposal = f + .directory + .prepare_node_bundle(&f.node, &frames, &[assigned], NOW) + .await + .unwrap(); + let (node, _) = f + .directory + .select_node_bundle(&f.node, &proposal, &f.lease, Limits::default(), NOW) + .await + .unwrap(); + f.node = node; + let checkpoints: Vec<_> = cells + .iter() + .map(|cell| { + let proof = proofs + .iter() + .find(|proof| proof.binding() == cell.control.value().bundle_binding.unwrap()) + .unwrap(); + (&cell.authority, proof) + }) + .collect(); + f.count.reset(); + f.node = f + .directory + .checkpoint_bundle_cells(&f.node, &checkpoints, Limits::default(), NOW) + .await + .unwrap(); + assert_eq!( + f.count.put_requests(), + 2, + "64 root checkpoints share one index upload and one CAS" + ); + for (number, cell) in cells.iter().enumerate() { + let current = cell + .authority + .load(cell.control.value().cell) + .await + .unwrap() + .unwrap(); + let proof = f + .directory + .load_bundle_coverage(&cell.authority, ¤t, Limits::default()) + .await + .unwrap(); + assert_eq!(proof.base().unwrap().commit_sequence, 2); + assert_eq!(proof.commit_sequence(), if number == 0 { 3 } else { 2 }); + assert_eq!( + proof.locator_count(), + if number == 0 { frames.len() } else { 0 } + ); + } +} + +#[tokio::test] +async fn checkpoint_cohort_refuses_duplicates_and_unmaterialized_participants_without_partial_cas() +{ + let mut f = Fixture::new().await; + let mut a = f.cell(4).await; + let mut b = f.cell(5).await; + let (_, mut frames, a_range) = f.append(&mut a, 2); + let (_, other, b_range) = f.append(&mut b, 2); + frames.extend(other); + let proposal = f + .directory + .prepare_node_bundle(&f.node, &frames, &[a_range, b_range], NOW) + .await + .unwrap(); + let (node, proofs) = f + .directory + .select_node_bundle(&f.node, &proposal, &f.lease, Limits::default(), NOW) + .await + .unwrap(); + f.node = node; + let a_proof = proofs + .iter() + .find(|proof| proof.binding() == a.control.value().bundle_binding.unwrap()) + .unwrap(); + let b_proof = proofs + .iter() + .find(|proof| proof.binding() == b.control.value().bundle_binding.unwrap()) + .unwrap(); + f.publisher(&a).materialize_bundle(a_proof).await.unwrap(); + f.count.reset(); + assert!(matches!( + f.directory + .checkpoint_bundle_cells( + &f.node, + &[(&a.authority, a_proof), (&b.authority, b_proof)], + Limits::default(), + NOW + ) + .await, + Err(Error::PendingPublication) + )); + assert_eq!( + f.count.put_requests(), + 0, + "a failed cohort selects none of its roots" + ); + assert!(matches!( + f.directory + .checkpoint_bundle_cells( + &f.node, + &[(&a.authority, a_proof), (&a.authority, a_proof)], + Limits::default(), + NOW + ) + .await, + Err(Error::Fenced) + )); + assert!(matches!( + f.directory + .checkpoint_bundle_cells(&f.node, &[], Limits::default(), NOW) + .await, + Err(Error::Capacity(_)) + )); + assert_eq!(f.count.put_requests(), 0); + let selected = f + .directory + .load_bundle_coverage(&a.authority, &a.control, Limits::default()) + .await + .unwrap(); + assert_eq!( + selected.base().unwrap().commit_sequence, + 1, + "the first root is still uncheckpointed" + ); + assert!(selected.locator_count() > 0); +} diff --git a/crates/cellule-runtime/src/node/bundle/tests/index/compatibility.rs b/crates/cellule-runtime/src/node/bundle/tests/index/compatibility.rs new file mode 100644 index 00000000..21e6e14e --- /dev/null +++ b/crates/cellule-runtime/src/node/bundle/tests/index/compatibility.rs @@ -0,0 +1,172 @@ +use super::*; + +#[tokio::test] +async fn legacy_selected_catalog_is_read_and_migrated_without_changing_the_cell_pin() { + let mut f = Fixture::new().await; + let mut cell = f.cell(4).await; + let original = f.node.advertisement().bundle_head().unwrap(); + let mut catalog = load_catalog(&f.layout, head_session(&f), original) + .await + .unwrap(); + catalog.index = None; + catalog.predecessor = Some(original.digest()); + let body = codec::encode(&mut catalog, &[]).unwrap(); + let legacy = PreparedNodeBundle { + original: Some(original), + head: NodeBundleHead { + epoch: EPOCH, + digest: Digest::from_bytes(*blake3::hash(&body).as_bytes()), + selected_through: 0, + }, + body, + catalog, + }; + f.layout + .store() + .put_exact( + &f.layout + .node_coverage_bundle_path(&[1; 16], EPOCH, legacy.head.digest.as_bytes()), + legacy.body.clone(), + ) + .await + .unwrap(); + f.node = f + .directory + .select_catalog(&f.node, &legacy, NOW) + .await + .unwrap(); + let old_pin = cell.control.value().bundle_binding.unwrap(); + let loaded = f + .directory + .load_bundle_coverage(&cell.authority, &cell.control, Limits::default()) + .await + .unwrap(); + assert_eq!(loaded.binding(), old_pin); + assert_eq!(loaded.commit_sequence(), 1); + let (_, frames, assigned) = f.append(&mut cell, 2); + let proposal = f + .directory + .prepare_node_bundle(&f.node, &frames, &[assigned], NOW) + .await + .unwrap(); + assert_eq!(&proposal.body[..8], b"\0\0\0\x04CNB2"); + let (_, proofs) = f + .directory + .select_node_bundle(&f.node, &proposal, &f.lease, Limits::default(), NOW) + .await + .unwrap(); + assert_eq!(proofs[0].binding(), old_pin); + assert_eq!(proofs[0].commit_sequence(), 2); +} + +#[tokio::test] +async fn indexed_shard_corruption_and_missing_reused_shards_fail_closed() { + let mut f = Fixture::new().await; + let mut a = f.cell(4).await; + let mut b = f.cell(5).await; + assert_ne!( + catalog_index::shard( + a.authority.layout().application_id(), + a.control.value().cell.as_bytes() + ), + catalog_index::shard( + b.authority.layout().application_id(), + b.control.value().cell.as_bytes() + ) + ); + let (_, frames, assigned) = f.append(&mut a, 2); + let proposal = f + .directory + .prepare_node_bundle(&f.node, &frames, &[assigned], NOW) + .await + .unwrap(); + let (selected, _) = f + .directory + .select_node_bundle(&f.node, &proposal, &f.lease, Limits::default(), NOW) + .await + .unwrap(); + f.node = selected; + let old_path = + f.layout + .node_coverage_bundle_path(&[1; 16], EPOCH, proposal.head.digest.as_bytes()); + let (_, frames, assigned) = f.append(&mut b, 2); + let sibling = f + .directory + .prepare_node_bundle(&f.node, &frames, &[assigned], NOW) + .await + .unwrap(); + let (selected, _) = f + .directory + .select_node_bundle(&f.node, &sibling, &f.lease, Limits::default(), NOW) + .await + .unwrap(); + f.node = selected; + f.count.reset(); + assert_eq!( + f.directory + .load_bundle_coverage(&a.authority, &a.control, Limits::default()) + .await + .unwrap() + .commit_sequence(), + 2 + ); + let reads: Vec<_> = f + .count + .requests() + .into_iter() + .filter(|read| read.location.ends_with(".cnb")) + .collect(); + assert_eq!( + reads.len(), + 3, + "a reused shard does not require the old root header" + ); + assert_eq!( + reads + .iter() + .filter(|read| read.location == old_path.as_ref()) + .count(), + 2 + ); + // The head/header stays unchanged; the independently authenticated reused + // shard must still detect a corrupt provider payload in the old object. + let mut corrupted = proposal.body.to_vec(); + corrupted[catalog_index::HEADER_BYTES + 8] ^= 1; + f.layout + .store() + .put_overwrite(&old_path, Bytes::from(corrupted)) + .await + .unwrap(); + assert!( + f.directory + .load_bundle_coverage(&a.authority, &a.control, Limits::default()) + .await + .is_err() + ); + // Unrelated Cell data remains readable; complete maintenance inventory must + // refuse the missing/corrupt dependency rather than omitting this Cell. + assert_eq!( + f.directory + .load_bundle_coverage(&b.authority, &b.control, Limits::default()) + .await + .unwrap() + .commit_sequence(), + 2 + ); + assert!( + load_catalog( + &f.layout, + head_session(&f), + f.node.advertisement().bundle_head().unwrap() + ) + .await + .is_err() + ); + f.layout.store().delete(&old_path).await.unwrap(); + assert!( + f.directory + .load_bundle_coverage(&a.authority, &a.control, Limits::default()) + .await + .is_err() + ); +} diff --git a/crates/cellule-runtime/src/node/bundle/tests/index/copy_on_write.rs b/crates/cellule-runtime/src/node/bundle/tests/index/copy_on_write.rs new file mode 100644 index 00000000..0443ed43 --- /dev/null +++ b/crates/cellule-runtime/src/node/bundle/tests/index/copy_on_write.rs @@ -0,0 +1,57 @@ +use super::*; + +#[tokio::test] +async fn updating_one_of_a_thousand_cells_does_not_rewrite_the_complete_inventory() { + let mut f = Fixture::new().await; + let mut cell = f.cell(4).await; + inventory(&mut f, &cell, 1_000).await; + let (_, frames, assigned) = f.append(&mut cell, 2); + f.count.reset(); + let prepared = f + .directory + .prepare_node_bundle(&f.node, &frames, &[assigned], NOW) + .await + .unwrap(); + let frame_bytes: usize = frames.iter().map(|frame| frame.encoded().len()).sum(); + assert!( + prepared.body.len() <= frame_bytes + (64 << 10), + "one changed Cell rewrote {} metadata bytes", + prepared.body.len() - frame_bytes + ); + eprintln!( + "catalog update: cells=1000 metadata_bytes={} native_bytes={frame_bytes}", + prepared.body.len() - frame_bytes + ); + let (selected, proofs) = f + .directory + .select_node_bundle(&f.node, &prepared, &f.lease, Limits::default(), NOW) + .await + .unwrap(); + assert_eq!(proofs.len(), 1); + assert_eq!(f.count.put_requests(), 2, "indexing adds no extra PUT"); + f.node = selected; + f.count.reset(); + let recovered = f + .directory + .load_bundle_coverage(&cell.authority, &cell.control, Limits::default()) + .await + .unwrap(); + assert_eq!(recovered.commit_sequence(), 2); + let catalog_reads: Vec<_> = f + .count + .requests() + .into_iter() + .filter(|read| read.location.ends_with(".cnb")) + .collect(); + assert!( + catalog_reads + .iter() + .all(|read| read.kind == cellule_store::test_support::ObjectReadKind::Range), + "one-Cell lookup downloaded complete bundle objects: {catalog_reads:?}" + ); + assert_eq!( + catalog_reads.len(), + 3, + "one header, one shard and one exact native frame" + ); +} diff --git a/crates/cellule-runtime/src/node/bundle/tests/index/inventory.rs b/crates/cellule-runtime/src/node/bundle/tests/index/inventory.rs new file mode 100644 index 00000000..edb8deb6 --- /dev/null +++ b/crates/cellule-runtime/src/node/bundle/tests/index/inventory.rs @@ -0,0 +1,63 @@ +use super::*; + +#[tokio::test] +async fn streamed_maintenance_inventory_verifies_the_last_shard_before_accepting_closure() { + let mut f = Fixture::new().await; + let cell = f.cell(4).await; + inventory(&mut f, &cell, 1_000).await; + let head = f.node.advertisement().bundle_head().unwrap(); + let mut catalog = load_catalog(&f.layout, head_session(&f), head) + .await + .unwrap(); + // Synthetic terminal rows exercise catalog integrity only. No Cell release, + // root ownership, process closure or collection is granted by this fixture. + for binding in &mut catalog.bindings { + binding.phase = BindingPhase::Closed; + binding.terminal = Some(( + binding.selected_sequence, + binding.selected_commit, + binding.selected_position, + )); + } + let proposal = f + .directory + .upload_catalog(Some(head), catalog, &[]) + .await + .unwrap(); + f.node = f + .directory + .select_catalog(&f.node, &proposal, NOW) + .await + .unwrap(); + f.count.reset(); + store::ensure_session_drained(&f.layout, head_session(&f), Some(proposal.head)) + .await + .unwrap(); + let reads: Vec<_> = f + .count + .requests() + .into_iter() + .filter(|read| read.location.ends_with(".cnb")) + .collect(); + assert!(reads.len() > 240, "closure checks every nonempty shard"); + assert!( + reads + .iter() + .all(|read| read.kind == cellule_store::test_support::ObjectReadKind::Range) + ); + let mut corrupted = proposal.body.to_vec(); + *corrupted.last_mut().unwrap() ^= 1; + let path = f + .layout + .node_coverage_bundle_path(&[1; 16], EPOCH, proposal.head.digest.as_bytes()); + f.layout + .store() + .put_overwrite(&path, Bytes::from(corrupted)) + .await + .unwrap(); + assert!( + store::ensure_session_drained(&f.layout, head_session(&f), Some(proposal.head)) + .await + .is_err() + ); +} diff --git a/crates/cellule-runtime/src/node/bundle/tests/index/mod.rs b/crates/cellule-runtime/src/node/bundle/tests/index/mod.rs new file mode 100644 index 00000000..33e23864 --- /dev/null +++ b/crates/cellule-runtime/src/node/bundle/tests/index/mod.rs @@ -0,0 +1,46 @@ +use super::*; +use crate::node::bundle::index as catalog_index; + +/// The other entries deliberately have no available origin roots. Selection +/// and lookup of this Cell must not open unrelated reconstruction dependencies. +async fn inventory(f: &mut Fixture, cell: &Cell, count: usize) { + let head = f.node.advertisement().bundle_head().unwrap(); + let mut catalog = load_catalog(&f.layout, head_session(f), head) + .await + .unwrap(); + let seed = catalog.bindings[0].clone(); + for number in 1..count { + let mut row = seed.clone(); + let mut id = [0_u8; 32]; + id[..8].copy_from_slice(&(number as u64).to_le_bytes()); + row.control.cell = CellId::from_bytes(id); + row.control.bundle_binding.as_mut().unwrap().digest = + Digest::from_bytes(*blake3::hash(&id).as_bytes()); + catalog.bindings.push(row); + } + catalog + .bindings + .sort_unstable_by_key(|row| *row.control.bundle_binding.unwrap().digest.as_bytes()); + let prepared = f + .directory + .upload_catalog(Some(head), catalog, &[]) + .await + .unwrap(); + f.node = f + .directory + .select_catalog(&f.node, &prepared, NOW) + .await + .unwrap(); + assert_eq!(cell.control.value().cell, CellId::from_bytes([4; 32])); +} + +fn head_session(f: &Fixture) -> SessionId { + f.node.advertisement().session() +} + +mod bootstrap; +mod checkpoint; +mod cohort; +mod compatibility; +mod copy_on_write; +mod inventory; diff --git a/crates/cellule-runtime/src/node/bundle/tests/lifecycle.rs b/crates/cellule-runtime/src/node/bundle/tests/lifecycle.rs index 968de53b..8455379f 100644 --- a/crates/cellule-runtime/src/node/bundle/tests/lifecycle.rs +++ b/crates/cellule-runtime/src/node/bundle/tests/lifecycle.rs @@ -53,11 +53,10 @@ async fn verify_hot_materialization(checkpoint_first: bool) { .await .unwrap(); assert_eq!(between.commit_sequence(), 3); - let pin = cell.control.value().bundle_binding.unwrap(); let (node, proof) = if checkpoint_first { let checkpoint = f .directory - .checkpoint_bundle_cell(&latest, &cell.authority, pin, Limits::default(), NOW) + .checkpoint_bundle_cell(&latest, &cell.authority, &old, Limits::default(), NOW) .await .unwrap(); let current = cell @@ -88,7 +87,7 @@ async fn verify_hot_materialization(checkpoint_first: bool) { ); let checkpoint = f .directory - .checkpoint_bundle_cell(&node, &cell.authority, pin, Limits::default(), NOW) + .checkpoint_bundle_cell(&node, &cell.authority, &proof, Limits::default(), NOW) .await .unwrap(); let current = cell @@ -129,10 +128,15 @@ async fn checkpoint_releases_locators_and_the_next_range_continues_exactly() { .select_node_bundle(&f.node, &prepared, &f.lease, Limits::default(), NOW) .await .unwrap(); - let pin = cell.control.value().bundle_binding.unwrap(); assert!(matches!( f.directory - .checkpoint_bundle_cell(&selected, &cell.authority, pin, Limits::default(), NOW) + .checkpoint_bundle_cell( + &selected, + &cell.authority, + &proofs[0], + Limits::default(), + NOW + ) .await, Err(Error::PendingPublication) )); @@ -147,7 +151,7 @@ async fn checkpoint_releases_locators_and_the_next_range_continues_exactly() { .unwrap(); f.node = f .directory - .checkpoint_bundle_cell(&selected, &cell.authority, pin, Limits::default(), NOW) + .checkpoint_bundle_cell(&selected, &cell.authority, &proof, Limits::default(), NOW) .await .unwrap(); let checkpoint = f @@ -197,12 +201,17 @@ async fn quiet_binding_must_close_and_checkpoint_before_session_withdrawal() { .unwrap(); let closed = f .directory - .finish_bundle_close(&closing, pin, NOW) + .finish_bundle_close(&closing, pin, issued, NOW) + .await + .unwrap(); + let proof = f + .directory + .load_bundle_coverage(&cell.authority, &cell.control, Limits::default()) .await .unwrap(); let checkpoint = f .directory - .checkpoint_bundle_cell(&closed, &cell.authority, pin, Limits::default(), NOW) + .checkpoint_bundle_cell(&closed, &cell.authority, &proof, Limits::default(), NOW) .await .unwrap(); f.directory.withdraw(&checkpoint, NOW).await.unwrap(); diff --git a/crates/cellule-runtime/src/node/bundle/tests/mod.rs b/crates/cellule-runtime/src/node/bundle/tests/mod.rs index 97372741..2848dcd5 100644 --- a/crates/cellule-runtime/src/node/bundle/tests/mod.rs +++ b/crates/cellule-runtime/src/node/bundle/tests/mod.rs @@ -16,6 +16,7 @@ use std::sync::Arc; const NOW: i64 = 1_000_000; const EPOCH: u64 = 2; mod faults; +mod index; mod lifecycle; mod ranges; struct Fixture { @@ -114,6 +115,14 @@ impl Fixture { cell } async fn unbound_cell_for_application(&mut self, byte: u8, application: [u8; 16]) -> Cell { + self.unbound_cell_at_commit(byte, application, 1).await + } + async fn unbound_cell_at_commit( + &mut self, + byte: u8, + application: [u8; 16], + commit_sequence: u64, + ) -> Cell { let layout = self.layout.for_application(application); let cell = CellId::from_bytes([byte; 32]); let incarnation = IncarnationId::from_bytes([byte + 10; 16]); @@ -134,7 +143,10 @@ impl Fixture { Limits::default(), ) .unwrap(); - let prepared = replica.prepare(None, &cuts, 1, 1).await.unwrap(); + let prepared = replica + .prepare(None, &cuts, commit_sequence, 1) + .await + .unwrap(); let mut control = Control::initial( cell, incarnation, @@ -379,7 +391,9 @@ async fn closure_drains_prior_fleet_ack_rejects_old_cas_and_blocks_transfer_unti .is_err() ); assert!(matches!( - f.directory.finish_bundle_close(&closing, pin, NOW).await, + f.directory + .finish_bundle_close(&closing, pin, issued, NOW) + .await, Err(Error::PendingPublication) )); let successor = a @@ -408,7 +422,7 @@ async fn closure_drains_prior_fleet_ack_rejects_old_cas_and_blocks_transfer_unti .unwrap(); let closed = f .directory - .finish_bundle_close(&selected, pin, NOW) + .finish_bundle_close(&selected, pin, issued, NOW) .await .unwrap(); assert!(matches!( @@ -438,7 +452,7 @@ async fn closure_drains_prior_fleet_ack_rejects_old_cas_and_blocks_transfer_unti // would retain locators that the departed writer can no longer release. let closed = f .directory - .checkpoint_bundle_cell(&closed, &a.authority, pin, Limits::default(), NOW) + .checkpoint_bundle_cell(&closed, &a.authority, &proofs[0], Limits::default(), NOW) .await .unwrap(); let released = a diff --git a/crates/cellule-runtime/src/node/bundle/tests/ranges.rs b/crates/cellule-runtime/src/node/bundle/tests/ranges.rs index 6f947137..1759e083 100644 --- a/crates/cellule-runtime/src/node/bundle/tests/ranges.rs +++ b/crates/cellule-runtime/src/node/bundle/tests/ranges.rs @@ -288,6 +288,35 @@ async fn locator_pressure_refuses_selection_without_dropping_the_last_proof() { .commit_sequence(), last.commit_sequence() ); + f.count.reset(); + let root = f.publisher(&cell).materialize_bundle(&last).await.unwrap(); + eprintln!( + "bounded suffix materialization: locators={} puts={}", + MAX_LOCATORS, + f.count.put_requests() + ); + assert_eq!( + f.count.put_requests(), + 4, + "small retained suffix uses one canonical coalesced pack" + ); + let restored = f.scratch.path().join("locator-pressure-restored.sqlite"); + cell.replica + .open_root(&root) + .await + .unwrap() + .restore(&restored) + .await + .unwrap(); + let db = rusqlite::Connection::open(restored).unwrap(); + let count: usize = db + .query_row("SELECT count(*) FROM outcomes", [], |row| row.get(0)) + .unwrap(); + assert_eq!( + count, + MAX_LOCATORS + 1, + "the later unselected command is absent" + ); } #[tokio::test] diff --git a/docs/bundle-coverage-implementation.md b/docs/bundle-coverage-implementation.md index cc62a65d..ffc0f53b 100644 --- a/docs/bundle-coverage-implementation.md +++ b/docs/bundle-coverage-implementation.md @@ -6,7 +6,8 @@ enabled in the ordinary actor response path**. The prior performance regression and failed qualification remain the baseline. This slice establishes ordering and reconstruction evidence. The [fresh application-path benchmark](pr67-performance-reevaluation.md) measures `7fc0793`; it does not exercise bundle-based responses or establish -write parity. +write parity. The [WAL NORMAL comparison](pr67-normal-wal-reevaluation.md) +separately records seven completed diagnostic cases and an interrupted matrix. ## Celld reference and Cellule adaptation @@ -34,9 +35,9 @@ Uploading an immutable object supplies no coverage proof. | `prepare_node_bundle` / `select_node_bundle` | Reject missing, overlapping, unassigned or cross-binding ranges; verify origin extents before selecting; reconcile only the exact head under the original lease | | `BundleCoverageProof` | Retain authenticated immutable locators, logical endpoint, SQLite position and original base; retain no capture bodies | | `load_bundle_coverage` | Reopen a selected suffix from the authority-pinned canonical catalog, including a fenced boot; grant no writer or follower-suffix closure | -| `CellPublisher::materialize_bundle` | Reconstruct the exact overlay and use normal root preparation, lineage and Cell CAS independently of selection | -| `checkpoint_bundle_cell` | Read current Cell authority and drop only an exact complete materialized prefix; retain newer selected suffix locators | -| `close_cell_issuance` / `begin_bundle_close` / `finish_bundle_close` | Freeze complete assigned issuance, including prior Fleet ACKs above the selected prefix; permit only its old issued tail to drain; close only at the exact terminal endpoint | +| `CellPublisher::materialize_bundle` | Reconstruct the exact overlay and use normal root preparation, lineage and Cell CAS independently of selection; bounded small tails reuse canonical native coalescing and packing | +| `checkpoint_bundle_cell` / `checkpoint_bundle_cells` | Require the opaque proof matching each freshly read materialized root; drop only its identical locator prefix, retain newer suffixes, and select up to 64 checkpoints with one index upload and one node CAS | +| `close_cell_issuance` / `begin_bundle_close` / `finish_bundle_close` | Freeze complete assigned issuance, including prior Fleet ACKs above the selected prefix; retain that same opaque issuance witness through terminal closure; read only its authenticated shard | | Cell departure CAS | Refuse release, takeover and tombstone until Closed, exact materialization and catalog checkpoint; migration refuses while bound | | Node withdrawal/maintenance | Refuse unresolved bindings; stale fencing preserves the catalog head; one bundle-bound boot cannot rotate its native log to another epoch | | Backup and collection | Backup refuses bound Cells. Coverage objects have no deletion path; this is retention, not a qualified collection implementation | @@ -75,15 +76,31 @@ unknown fields are rejected. Old readers must not operate on those records. Use a fresh development prefix and initialize the lane before native issuance; there is no live-format migration or same-boot bundle-epoch rotation. -The `CNB1` object embeds a complete catalog and native frame extents under -`cells/v1/node-logs///coverage/v1/.cnb`. It is outside the -advertisement scan prefix. Both manifests and individual frame extents are -authenticated. New-object locators use a canonical self reference, resolved -against the selected digest, avoiding a self-referential digest field. +New `CNB2` objects contain a fixed authenticated root index, changed catalog +shards and native extents under the existing +`cells/v1/node-logs///coverage/v1/.cnb` path. The selected +head digest authenticates the 32 KiB canonical header; the header authenticates +all shard ranges and new native extents. Unchanged shard references retain their +original object, offset, length and digest. A Cell lookup needs one header and +one shard, then its exact native ranges; it never follows a predecessor chain. +Both the application and Cell ID determine the shard. Complete maintenance +inventory still reads and verifies every shard, retaining one bounded shard at +a time and a duplicate-pin set capped at 4,096 entries. +Hot cohort lookups preflight the sum of their chosen shard lengths against a +4 MiB encoded-metadata budget before leaf I/O; a point lookup charges only its +own shard. This bounds protocol input, not total heap usage or host admission. + +The `CNB1` complete-catalog reader remains available. Its selected digest still +authenticates the whole object. A new update migrates that catalog to `CNB2` +without changing Cell pins. Older binaries cannot read `CNB2`; upgrade all +recovery consumers before selecting this format. This does not migrate unbound +live actor activations or enable bundle-based responses. | Development bound | Value | | --- | ---: | -| Immutable object/catalog bytes | 4 MiB | +| Immutable object or individual catalog shard bytes | 4 MiB | +| Encoded catalog shards loaded for one cohort | 4 MiB, checked before leaf reads | +| Authenticated index header | 32 KiB / 256 shard references | | Original bindings per boot/epoch | 4,096 | | Native frames per selection | 64 | | Uncheckpointed locators per Cell | 32 | @@ -91,16 +108,47 @@ against the selected digest, avoiding a self-referential digest field. | Distinct base-root origin dependencies verified per Cell | 65,536 | Exhaustion rejects further bundle preparation while retaining the last proof. -These are safety ceilings, **not performance-qualified policies**. The complete -catalog is rewritten per selection and old locator bodies are reverified. -This is not yet the file-backed index needed for the 160/214-command checkpoint -cost constraints in the [authority decision](bundle-coverage-proof.md#quantified-checkpoint-constraint). +These are safety ceilings, **not performance-qualified policies**. New selections rewrite only changed shards. Old locator bodies are still +reverified. The indexed point lookup and copy-on-write catalog are implemented; +the 32-locator bound, materializer policy and full lifecycle cost do not yet meet +the current conditional 215-command checkpoint cost constraint in the [authority decision](bundle-coverage-proof.md#quantified-checkpoint-constraint). The caller owns host memory/native-job admission; automatic materializer budget, fairness and cancellation/drain ownership are not integrated. ## Verification and remaining delivery -Twenty-three focused tests cover real managed SQLite captures, canonical object-store +The index regression reproduces a **783,146-byte** metadata rewrite for one +update among 1,000 Cells on the old source. The same test measures **35,223 bytes** +with `CNB2` (95.5% less), retains one immutable PUT plus one node CAS, and recovers +through three coverage-object ranges: header, chosen shard and native frame. +An exact materialized-prefix checkpoint among those 1,000 Cells falls from +**253 coverage-object reads to two**, without reading old native frames. A +64-Cell checkpoint cohort selects all completed roots with **two PUTs**, keeping +a hot Cell's newer suffix; an unmaterialized participant rejects the whole cohort +without a partial CAS. Small independent materialization now uses the canonical +native coalescer and pack: **256 PUTs instead of 320** for those 64 roots, or four +per root. A 32-locator suffix falls from **37 PUTs to four**, and cold restore +includes every selected outcome while excluding the later unselected command. +The aggregate native rows, indexes and pack headers must fit the existing +256 KiB budget; larger tails preserve the ordinary bundle representation. +The [updated cost calculation](bundle-coverage-proof.md#quantified-checkpoint-constraint) +requires at least 215 commands per Cell checkpoint if that four-PUT path and +64/64 cohorts hold, before retries or maintenance. Real larger checkpoints must +be measured; the 32-locator safety ceiling does not satisfy this constraint. + +Additional tests cover legacy migration with the same Cell pin, a reused shard +without reading its old header, corrupt and missing reused shards, and refusal +to omit a corrupt shard from complete inventory. Bootstrap enrollment accepts +the runtime's command-zero root only before any native issuance; the first +command materializes and a quiet Cell drains without inventing a command. +An exact installed checkpoint retry preserves hot newer locators without another +PUT. A chosen cohort exceeding the metadata budget refuses before fetching its +first leaf. These regressions were reproduced before their fixes. +This is metadata/I/O evidence, +not application TPS or complete M4 qualification. + + +The original twenty-three focused tests cover real managed SQLite captures, canonical object-store CAS and materialization: two-Cell shared selection, byte-identical roots, a prior Fleet ACK above the selected prefix, held old proposals across closure, hot siblings and dormant bindings, lost replies, lease loss, partial command groups, @@ -122,19 +170,36 @@ with both Cell roots unchanged. That excludes enrollment, root materialization, catalog checkpoints, compaction and maintenance and must not be reported as total PUTs/command, TPS or a latency result. -The isolated source snapshot passed all contributor checks: **1,887 workspace +The original bundle source snapshot passed all contributor checks: **1,887 workspace tests passed, 38 ignored; 58 local LTX tests passed**. All-feature/all-target checking, warnings-denied Clippy and API docs, formatting, boundaries, module ownership, document syntax/links and SQL/peer validation also passed. Ignored environment-dependent tests and the complete production qualification remain outside this result. +The final indexed publication/materialization snapshot passed all twelve +contributor checks: **1,902 workspace tests passed, 38 ignored; 60 local LTX +tests passed**. This includes warnings-denied Clippy and API docs, all-feature +and all-target checking, formatting, boundary/layout checks, document syntax +and links, SQL/peer validation and performance-harness tests. Its 34 focused +bundle/index tests and two public recovery-overlay tests cover the new cost, +checkpoint retry, command-zero bootstrap and larger-tail fallback behavior. +Source manifests, failed-before/pass-after logs and build outputs remain in the +external `cellule-write-perf-8ad1` evidence directory. Environment-dependent +ignored tests and production performance qualification remain outstanding. + +The experimental checkpoint API now takes its `BundleCoverageProof`, and +`finish_bundle_close` takes the same retained `CellIssuedRange` used for Closing. +Consumers must pass those original capabilities; a pin or sampled endpoint is +not a substitute. The singleton checkpoint method delegates to the cohort path. + Remaining work before responses can use bundle proof: 1. Integrate actor/executor proof and query/retry endpoints, release proved capture retention, and schedule admitted materializers with joined shutdown. -2. Replace complete catalog rewrites with a bounded authenticated file-backed - lookup/checkpoint index; measure the full checkpoint and collection cost. +2. Build on the authenticated index: bound admitted maintenance inventory, + increase checkpoint density with retained-byte accounting, and measure the + complete materialization/checkpoint/collection cost. 3. Fence and seal a failed original node's complete follower-issued suffix, including ACKs above the selected head, before reconstructing and closing every original binding, including interrupted provisional enrollments. diff --git a/docs/bundle-coverage-proof.md b/docs/bundle-coverage-proof.md index c02258b4..1feda875 100644 --- a/docs/bundle-coverage-proof.md +++ b/docs/bundle-coverage-proof.md @@ -181,6 +181,29 @@ cannot release those payload objects. Account for lookup checkpoints and actual data reclamation separately; independent checkpoints cannot be treated as free uploads. +The first indexed prototype provided a measured correction to those optimistic +bounds. Its 64-Cell materialization test observed **320 successful PUTs**, or +five per root, while the new shared checkpoint adds two PUTs for the entire +64-Cell cohort. Keeping both selection and checkpoint cohorts at 64 gives +`2 / 64 + (5 + 2 / 64) / commands_per_checkpoint <= 0.05`: **at least 269 commands +per Cell checkpoint**, before compaction, retries or collection. With an actual +three-PUT materializer that bound would be 162; with four it would be 215 once +the shared checkpoint is included. These are conditional calculations, not a +qualified application result. At one command per capture, the current +32-locator limit cannot meet even the optimistic bound. Raising that limit +without admitted file-backed metadata and bounded cold/read cost is insufficient. + +The subsequent canonical-pack optimization reduces the same 64-root cohort to +**256 PUTs** and a 32-locator suffix from **37 PUTs to four**. Independent tails +use this path only when their aggregate native rows, indexes and pack headers +fit the existing 256 KiB budget; larger tails retain the bundle representation. +If that four-PUT materialization holds at the eventual checkpoint spacing, +`2 / 64 + (4 + 2 / 64) / commands_per_checkpoint <= 0.05` requires **215 commands +per Cell checkpoint**, before compaction, retries or collection. This remains +conditional: checkpoint density, larger-tail cost and all-ACK cold/read bounds +must be measured after the file-backed locator work. The present 32-locator +ceiling still cannot satisfy it for one-command captures. + At the uniform 1,000-Cell bucket target of 2,000 commands/s, 160 commands per Cell span approximately 80 seconds; at 15,000 Fleet commands/s they span about 10.7 seconds. These are inferred cost constraints, not measured performance or diff --git a/docs/pr67-normal-wal-reevaluation.md b/docs/pr67-normal-wal-reevaluation.md new file mode 100644 index 00000000..0e20253c --- /dev/null +++ b/docs/pr67-normal-wal-reevaluation.md @@ -0,0 +1,85 @@ +# PR67 WAL NORMAL verification + +Runtime-managed SQLite activations opt into WAL `NORMAL` at +`075b2cd45cb2762f9795aab643e0a6ad14e4ca8d`. Standalone LTX databases keep +`FULL`. Command responses still require selected object publication or an exact +recoverable follower proof. Follower log fsync is unchanged. + +This **interrupted diagnostic comparison does not qualify write parity**. +Seven of nine planned cases completed. The FULL Bucket case reached warm audit +and clean owner exit, then the Docker VM stopped before cold audit and its final +case report. NORMAL Bucket did not start. The cause of the VM stop is unknown; +the original plan and incomplete evidence remain retained outside Git. + +## Matched workload and completed measurements + +The controls are Cellule `7fc079345b5214a18a001833732a84039e4485db` (FULL), +Cellule `075b2cd45cb2762f9795aab643e0a6ad14e4ca8d` (NORMAL), and celld +`f2bf648663a610eefde71f3547ad61e9b896b1f0`. Fresh pinned release builds used +identical clients and auditors. Each case used 1,000 uniformly selected Cells, +96-byte values, INSERT plus SELECT, durable request/result records, 128 clients, +a 30-second warmup and a 300-second offered-load window. This is not a bounded +KV overwrite workload. + +An ARM64 Linux Docker VM shared 8 CPUs and 16 GiB across the owner, two followers, +object store and load generator. Each node had 4 GiB tmpfs scratch. Cellule's +retention budgets were 64 MiB memory and 1 GiB managed disk. This is a simulation, +not three dedicated machines, and only one repetition of each point completed. + +| Durability / offered load | FULL completed TPS / p99 ms | NORMAL completed TPS / p99 ms | celld completed TPS / p99 ms | +| --- | ---: | ---: | ---: | +| Fleet / 100 per second | 98.18 / 601.7 | 99.91 / 369.7 | 100.00 / 56.8 | +| Fleet / 15,000 per second | 278.30 / 138.9 | 299.05 / 144.3 | 1,149.13 / 241.7 | +| Bucket / 2,000 per second | Interrupted: 121.43 / 9,585.6 | Not run | 499.76 / 2,234.8 | + +TPS counts successful requests completed inside the measurement window. p99 +covers scheduled latency of attempted requests, including errors; dropped +requests have no response latency. None of the completed cases passed the full +throughput/latency/delivery qualification. High-load completed TPS is not +sustainable capacity. The single Fleet pair's 7.5% NORMAL improvement needs +repetition; it does not establish a causal or repeatable end-to-end gain. + +## Recovery, pressure and publication evidence + +| Case | Verification result | +| --- | --- | +| All three Fleet 100 cases | Every ACK passed exact warm and cold retry; FULL had 384 errors and 161 drops, NORMAL 26 errors, celld zero | +| NORMAL Fleet 15K | 105,279 ACKs; warm audit had 1,349 HTTP 503s; drain exceeded 120 seconds; cold audit not reached | +| FULL Fleet 15K | 94,815 ACKs; warm audit had 59,092 HTTP 503s; provider later exited OOM 137, 43 seconds after final metrics; cold audit not reached | +| celld Fleet 15K | 481,602 ACKs; warm audit returned HTTP 500 throughout; live filesystem observation confirmed 100% tmpfs usage and logs reported no space left; cold audit not reached | +| celld Bucket 2K | 166,564 ACKs all passed warm retry; cold audit had 33 HTTP 500s with object-store LIST timeouts; provider remained healthy | +| Interrupted FULL Bucket 2K | 43,708 ACKs all passed warm retry; original owner exited zero; no completed cold audit or case report | + +An HTTP audit failure does not prove acknowledged data loss. The failed cold +and drain gates remain failures, not evidence to discard or rerun invisibly. + +At Fleet 15K, publication averaged 4,028 ms (FULL) and 4,088 ms (NORMAL), measured +from counter deltas. NORMAL ended with 72.0 MB of unpublished debt and an oldest +publication age of 166 seconds. At Fleet 100, successful publication PUTs per +logical command were 4.93 and 4.81 respectively. These measurements point to +publication/backlog work beyond SQLite synchronization. They exclude provider +SDK-internal retries and do not replace the full M4 cost accounting. + +## Isolated SQLite mechanism probe + +A separate local probe measured transaction plus commit observation only. It +excluded capture, proof, publication, HTTP and request latency. Every run then +passed exact LTX capture/restore. Three alternating 20,000-transaction pairs +produced these medians: + +| Scratch | FULL TPS / transaction p99 µs | NORMAL TPS / transaction p99 µs | +| --- | ---: | ---: | +| tmpfs | 44,768 / 331.2 | 50,756 / 340.5 | +| Docker Linux filesystem volume | 1,582 / 2,513.3 | 70,683 / 62.3 | + +These short probes establish the synchronization mechanism, not sustained +framework capacity or physical power-loss durability. A separate traced probe +observed 1,028 fsync calls with FULL and 25 with NORMAL across 1,000 transactions +plus setup/capture/close. Those are total calls, not exact per-command counts. + +The isolated NORMAL source passed all contributor checks: 1,890 workspace tests +passed, 38 ignored, and 60 local LTX tests passed. Raw logs, ACK manifests, +source/build hashes, counters and the interrupted matrix are retained under the +external `cellule-write-perf-8ad1` evidence directory. Qualification still requires +three matched runs with all-ACK cold recovery, stable debt, drain and read gates +from the [performance proposal](write-performance-proposal.md). diff --git a/docs/write-performance-delivery.md b/docs/write-performance-delivery.md index 9a253c02..fc9c44b2 100644 --- a/docs/write-performance-delivery.md +++ b/docs/write-performance-delivery.md @@ -14,6 +14,15 @@ modestly better, and provider/recovery failures remain explicitly recorded. The older measurements below are historical and are not measurements of the latest bundle APIs. +The [WAL NORMAL reevaluation](pr67-normal-wal-reevaluation.md) records a later +interrupted comparison at `075b2cd`; it does not qualify parity. The +[indexed bundle implementation](bundle-coverage-implementation.md) reduces one +1,000-Cell catalog update from 783,146 to 35,223 metadata bytes and checkpoints +64 exact roots with two shared PUTs. Those are protocol component measurements; +ordinary actor responses still use the previous publication path. +Canonical small-tail materialization also reduces the 64-root cohort from 320 +to 256 PUTs and a 32-locator suffix from 37 PUTs to four, with exact cold restore. + ## Delivered behavior | Change | Measurable result | Preserved contract | @@ -48,7 +57,7 @@ collection paths. There is no legacy decoding or automatic migration. | M1 | Packs, inline leaves and bounded compaction spooling delivered | Ordinary append meets four PUTs; composed compaction needs five. Paired cold/sparse-read guardrail unverified | | M2 | Bounded file-backed shared publication coordinator implemented; exact scope, restore, cancellation, minimum-budget and dormant-sibling retention checks added | Latest three active-Fleet windows cost 5.229–5.433 PUTs/command; two fail the debt trend. Three per-Cell authority PUTs remain; M4 is required | | M3 | Signed 512-sequence/five-second grants, bounded local registry, lifecycle gate and signed HTTP fixture implemented | Full isolated checks and native lifecycle suite pass; latest three active-Fleet windows cost 0.0138 enrollment GETs/command. The 15K target diagnostic fails delivery and warm audit | -| M4 | Models and [connected protocol APIs](bundle-coverage-implementation.md) for shared selection, exact assigned ranges, independent materialization, checkpoint and complete live-writer closure | Actor response/read integration, failed-node issued-suffix recovery, bounded index/host admission and bundle collection remain incomplete; bundle ACKs disabled | +| M4 | [Connected protocol APIs](bundle-coverage-implementation.md), authenticated copy-on-write catalog shards, exact prefix/cohort checkpoints and streamed complete inventory | Actor response/read integration, admitted materializer scheduling, checkpoint density, failed-node issued-suffix recovery and bundle collection remain incomplete; bundle ACKs disabled | | M5 | Three paired low-rate Fleet repetitions and target diagnostics with exact ACK audits delivered | Publication stability and target delivery fail; qualified capacity, read/failure/overload matrix and absolute/relative parity remain unverified | ## Shared publication checkpoint From 6e4ade6124aaafffae030023fdbd26f277bc922b Mon Sep 17 00:00:00 2001 From: forhappy Date: Wed, 7 Oct 2026 17:03:13 -0700 Subject: [PATCH 020/102] Retain dense authenticated histories and stream recovered materialization --- crates/cellule-ltx/docs/publication.md | 10 +- crates/cellule-ltx/src/bundle.rs | 15 +- crates/cellule-ltx/src/replica/coalesce.rs | 82 +++++- crates/cellule-ltx/src/replica/prepare.rs | 15 + .../cellule-ltx/tests/cell/roots/lifecycle.rs | 85 ++++++ .../docs/technical-reference.md | 1 + .../docs/write-performance-design.md | 104 +++++++ .../src/node/bundle/binding.rs | 11 +- .../src/node/bundle/closure.rs | 32 +- .../cellule-runtime/src/node/bundle/codec.rs | 5 +- .../src/node/bundle/index/codec.rs | 15 +- .../src/node/bundle/index/dense.rs | 276 ++++++++++++++++++ .../src/node/bundle/index/history.rs | 201 +++++++++++++ .../src/node/bundle/index/io.rs | 122 +++++++- .../src/node/bundle/index/mod.rs | 207 +++---------- .../src/node/bundle/index/tests.rs | 70 ----- .../src/node/bundle/index/tests/admission.rs | 169 +++++++++++ .../src/node/bundle/index/tests/histories.rs | 65 +++++ .../src/node/bundle/index/tests/legacy.rs | 172 +++++++++++ .../src/node/bundle/index/tests/mod.rs | 71 +++++ crates/cellule-runtime/src/node/bundle/mod.rs | 4 +- .../cellule-runtime/src/node/bundle/proof.rs | 11 +- .../src/node/bundle/selection.rs | 33 ++- .../cellule-runtime/src/node/bundle/store.rs | 28 +- .../src/node/bundle/tests/index/checkpoint.rs | 4 +- .../node/bundle/tests/index/compatibility.rs | 82 +++++- .../node/bundle/tests/index/copy_on_write.rs | 4 +- .../src/node/bundle/tests/index/density.rs | 210 +++++++++++++ .../src/node/bundle/tests/index/mod.rs | 1 + .../src/node/bundle/tests/ranges.rs | 13 +- docs/bundle-coverage-implementation.md | 111 +++++-- docs/bundle-coverage-proof.md | 29 +- docs/write-performance-delivery.md | 10 +- 33 files changed, 1890 insertions(+), 378 deletions(-) create mode 100644 crates/cellule-runtime/docs/write-performance-design.md create mode 100644 crates/cellule-runtime/src/node/bundle/index/dense.rs create mode 100644 crates/cellule-runtime/src/node/bundle/index/history.rs delete mode 100644 crates/cellule-runtime/src/node/bundle/index/tests.rs create mode 100644 crates/cellule-runtime/src/node/bundle/index/tests/admission.rs create mode 100644 crates/cellule-runtime/src/node/bundle/index/tests/histories.rs create mode 100644 crates/cellule-runtime/src/node/bundle/index/tests/legacy.rs create mode 100644 crates/cellule-runtime/src/node/bundle/index/tests/mod.rs create mode 100644 crates/cellule-runtime/src/node/bundle/tests/index/density.rs diff --git a/crates/cellule-ltx/docs/publication.md b/crates/cellule-ltx/docs/publication.md index ee327928..e69f9166 100644 --- a/crates/cellule-ltx/docs/publication.md +++ b/crates/cellule-ltx/docs/publication.md @@ -79,9 +79,13 @@ and the authority CAS continue to govern acknowledgement. Independent recovery uses the same path when all selected rows, their indexes and pack headers together fit the existing 256 KiB single-PUT budget. Every -original row is verified before coalescing. If a later row exceeds that budget, -earlier frozen bodies are released and the entire tail retains its ordinary -bundle representation. `prepare_bundle` preserves shared bundle references. +original row is verified before coalescing. If the aggregate exceeds that +budget, earlier frozen bodies are released. The fully validated original chain +then streams through the same coalescer, with one pinned source read per admitted +native job and the existing 256 KiB changed-page state. Long histories that +repeatedly change a small image can still produce one canonical pack. A large +individual row, changed image or output retains the ordinary bundle path. +`prepare_bundle` preserves shared bundle references. No root format, authority rule or host resource ceiling changes. A representation-only compaction can remain private while its successor append diff --git a/crates/cellule-ltx/src/bundle.rs b/crates/cellule-ltx/src/bundle.rs index 126c435d..e0d5821c 100644 --- a/crates/cellule-ltx/src/bundle.rs +++ b/crates/cellule-ltx/src/bundle.rs @@ -408,10 +408,21 @@ impl Bundle { /// Returns the verified bytes for a row index, never a caller-supplied extent. pub fn read_segment(&self, index: usize) -> Result { + self.segment_reader(index)?() + } + + /// Pins the immutable source for one admitted native read without retaining + /// every row's bytes. The dispatched closure owns file lifetime on cancellation. + pub(crate) fn segment_reader( + &self, + index: usize, + ) -> Result Result + Send + 'static> { let row = self.rows.get(index).ok_or(LtxError::TxNotAvailable)?; let start = usize::try_from(row.offset).map_err(|_| LtxError::LTXCorrupted)?; let length = usize::try_from(row.info.size_bytes).map_err(|_| LtxError::LTXCorrupted)?; - match &self.body { + let row = row.clone(); + let body = self.body.clone(); + Ok(move || match body { BundleBody::Memory(bytes) => Ok(bytes.slice(start..start + length)), BundleBody::File(file) => { let mut source = File::open(&file.path)?; @@ -422,7 +433,7 @@ impl Bundle { } Ok(Bytes::from(bytes)) } - } + }) } } diff --git a/crates/cellule-ltx/src/replica/coalesce.rs b/crates/cellule-ltx/src/replica/coalesce.rs index eb7041bf..57bd51dc 100644 --- a/crates/cellule-ltx/src/replica/coalesce.rs +++ b/crates/cellule-ltx/src/replica/coalesce.rs @@ -9,6 +9,60 @@ use super::{ use crate::{LtxError, Result, SegmentInfo, codec, ltx}; use bytes::Bytes; +/// Coalesces an independent recovered tail without freezing its entire input. +/// Every source row and index was admitted and the full chain validated first. +/// One source read plus the existing 256 KiB changed-page state crosses a native +/// job at a time. A large row or changed image keeps the original bundle path. +pub(super) async fn recovered_bundle( + replica: &CellReplica, + bundle: &crate::bundle::Bundle, + inputs: &[super::AppendInput], +) -> Result> { + if inputs.len() < 2 + || inputs.iter().any(|input| { + input.info.size_bytes > SINGLE_PUT_BYTES || input.index.len() as u64 > SINGLE_PUT_BYTES + }) + { + return Ok(None); + } + let (repository, epoch) = crate::bundle::cell_identity(&replica.cell, &replica.incarnation); + let rows: Vec<_> = bundle + .rows() + .iter() + .enumerate() + .filter(|(_, row)| row.repository == repository && row.epoch == epoch) + .collect(); + if rows.len() != inputs.len() { + return Err(LtxError::LTXCorrupted); + } + let mut state = MergeState::default(); + let mut original_bytes = 0_u64; + for ((row_index, row), input) in rows.into_iter().zip(inputs) { + if row.info != input.info + || !matches!(input.location, super::BodyLocation::Bundle { digest, offset } if digest == bundle.digest() && offset == row.offset) + { + return Err(LtxError::LTXCorrupted); + } + original_bytes = original_bytes + .checked_add(input.info.size_bytes) + .and_then(|bytes| bytes.checked_add(input.index.len() as u64)) + .ok_or(LtxError::LTXCorrupted)?; + let read = bundle.segment_reader(row_index)?; + let info = input.info.clone(); + let index = input.index.clone(); + let Some(merged) = replica + .host + .run(move || state.apply(read()?, &info, &index)) + .await + .map_err(pinned_storage_error)?? + else { + return Ok(None); + }; + state = merged; + } + finish(replica, state, original_bytes).await +} + pub(super) async fn run( replica: &CellReplica, segments: Vec, @@ -53,23 +107,33 @@ pub(super) async fn run( }; state = merged; } + match finish(replica, state, original_bytes).await? { + Some(merged) => Ok(vec![merged]), + None => Ok(segments), + } +} + +async fn finish( + replica: &CellReplica, + state: MergeState, + original_bytes: u64, +) -> Result> { let limits = replica.limits; - let merged = replica + let Some(merged) = replica .host .run(move || state.finish(limits.max_file_bytes.min(SINGLE_PUT_BYTES))) - .await??; - let Some(merged) = merged else { - return Ok(segments); + .await?? + else { + return Ok(None); }; if merged.descriptor.info.size_bytes + merged.index.len() as u64 > original_bytes { - return Ok(segments); + return Ok(None); } match replica.admit_segment_representation(&merged.descriptor.info, merged.index.len()) { - Ok(()) => {} - Err(LtxError::Limit(crate::LimitKind::CapturedCellLtxBytes)) => return Ok(segments), - Err(error) => return Err(error), + Ok(()) => Ok(Some(merged)), + Err(LtxError::Limit(crate::LimitKind::CapturedCellLtxBytes)) => Ok(None), + Err(error) => Err(error), } - Ok(vec![merged]) } #[derive(Default)] diff --git a/crates/cellule-ltx/src/replica/prepare.rs b/crates/cellule-ltx/src/replica/prepare.rs index b84ca7e5..177e2115 100644 --- a/crates/cellule-ltx/src/replica/prepare.rs +++ b/crates/cellule-ltx/src/replica/prepare.rs @@ -365,6 +365,21 @@ impl CellReplica { .map(|input| input.info.position()) .ok_or(LtxError::TxNotAvailable)?; self.validate_chain(&prospective, target)?; + if usage == BundleUse::IndependentRecovery && !independent { + // A long history can repeatedly update the same small page image. + // Its summed input exceeds a pack while its merged output fits. + // Stream original rows through the existing admitted coalescer; + // retain the bundle unchanged when the changed-image bound fails. + if let Some(merged) = coalesce::recovered_bundle(self, bundle, &inputs).await? { + inputs = vec![AppendInput { + info: merged.descriptor.info, + location: BodyLocation::Native, + index: merged.index, + body: merged.body, + }]; + independent = true; + } + } if independent { // Independent recovery already verified every original cut. The // canonical coalescer and native pack factory can now reduce their diff --git a/crates/cellule-ltx/tests/cell/roots/lifecycle.rs b/crates/cellule-ltx/tests/cell/roots/lifecycle.rs index 45424180..bad94eb6 100644 --- a/crates/cellule-ltx/tests/cell/roots/lifecycle.rs +++ b/crates/cellule-ltx/tests/cell/roots/lifecycle.rs @@ -880,3 +880,88 @@ async fn recovered_overlay_keeps_the_bundle_when_a_later_row_exceeds_the_pack_bu assert_eq!(values, vec![vec![0], vec![1], payload]); writer.close().unwrap(); } + +#[tokio::test] +async fn recovered_file_history_streams_more_than_a_pack_of_inputs_into_one_small_root() { + let directory = tempfile::tempdir().unwrap(); + let mut writer = Db::open(&directory.path().join("dense.sqlite"), Limits::default()).unwrap(); + writer.transaction(|tx| tx.execute_batch("CREATE TABLE outcomes(request TEXT PRIMARY KEY, result TEXT); INSERT INTO outcomes VALUES ('seed','original')")).unwrap(); + let cell = [183; 32]; + let incarnation = [184; 16]; + let counted = Arc::new(cellule_store::test_support::CountingObjectStore::new( + Arc::new(InMemory::new()), + )); + let replica = replica(Store::new(counted.clone()), cell, incarnation); + let base = replica + .prepare(None, &writer.capture().unwrap(), 1, 1) + .await + .unwrap() + .root(); + let mut builder = + cellule_ltx::bundle::BundleBuilder::new_temp(directory.path(), Limits::default()).unwrap(); + let mut total_bytes = 0_u64; + let mut position = base.position; + for commit in 2..=216 { + writer + .transaction(|tx| { + tx.execute( + "INSERT INTO outcomes VALUES (?1,?2)", + [format!("request-{commit}"), format!("result-{commit}")], + ) + }) + .unwrap(); + let cuts = writer.capture().unwrap(); + position = cuts.position; + for segment in &cuts.segments { + total_bytes += segment.info().size_bytes; + builder + .push(BundleEntry::for_cell( + cell, + incarnation, + segment.info().clone(), + std::fs::read(segment.path()).unwrap(), + )) + .unwrap(); + } + } + assert!( + total_bytes > 256 << 10, + "must exercise streamed fallback, not the small aggregate-input path: {total_bytes}" + ); + let overlay = RecoveryOverlay::new(base, builder.finish().unwrap(), position, 216); + counted.reset(); + let prepared = replica + .prepare_recovered_overlay(&overlay, 1) + .await + .unwrap(); + assert_eq!( + counted.put_requests(), + 2, + "one canonical pack and one root, excluding runtime lineage and CAS" + ); + assert_eq!(prepared.root().position, position); + assert_eq!(prepared.root().commit_sequence, 216); + let restored = directory.path().join("dense-restored.sqlite"); + replica + .open_root(&prepared.root()) + .await + .unwrap() + .restore(&restored) + .await + .unwrap(); + let db = rusqlite::Connection::open(restored).unwrap(); + let count: usize = db + .query_row("SELECT count(*) FROM outcomes", [], |row| row.get(0)) + .unwrap(); + assert_eq!(count, 216); + assert_eq!( + replica + .reachable_objects(&prepared.root()) + .await + .unwrap() + .iter() + .filter(|object| object.kind == CellObjectKind::Bundle) + .count(), + 0 + ); +} diff --git a/crates/cellule-runtime/docs/technical-reference.md b/crates/cellule-runtime/docs/technical-reference.md index f7c66b3a..6dfc5db9 100644 --- a/crates/cellule-runtime/docs/technical-reference.md +++ b/crates/cellule-runtime/docs/technical-reference.md @@ -21,6 +21,7 @@ flowchart LR | Framework topic | Reference | | --- | --- | | Runtime overview and ownership | [Understand the embedded Cell runtime](overview.md) | +| Node capacity, shared publication and performance exit gates | [Write and read performance design](write-performance-design.md) | | Actor, SQL worker, deadlines, and drain | [Execution and receipts](runtime.md) | | IDs, control, immutable roots, pages, backups, and retention | [Authority, storage, and recovery](storage.md) | | SQL, KV, Blob, Queue, Cron, Workflow, and Effects | [Cell primitives](primitives.md) | diff --git a/crates/cellule-runtime/docs/write-performance-design.md b/crates/cellule-runtime/docs/write-performance-design.md new file mode 100644 index 00000000..ab1e4f43 --- /dev/null +++ b/crates/cellule-runtime/docs/write-performance-design.md @@ -0,0 +1,104 @@ +# Node write and read performance design + +Status: implementation in progress. The application path is not qualified at +the targets below. Component I/O reductions are not application TPS. + +Cellule is the management process for a node's Cells. A Cell retains one fenced +writer; node-wide admission, native workers, publication, recovery and collection +must share bounded resources rather than create an independent service per Cell. +The embedding application continues to own ingress and authorization. + +## Capacity contract + +| Dimension | Required node result | +| --- | --- | +| Serving node | 8 vCPUs, 16 GiB memory; report storage, filesystem, network and SQLite policy | +| Population | 2,000 uniformly active Cells, sub-100-byte values | +| Writes | 10,000 successful commands/s | +| Reads | 50,000 successful queries/s | +| Tail latency | Fleet writes p99 at most 50 ms; Bucket writes p99 at most 200 ms; report read p50/p95/p99 | +| Delivery | Zero errors, dropped offers or unissued requests; at least 99% completed within the window | +| Evidence | Three paired repetitions of at least five minutes, matched celld revision and workload | +| Durability | All acknowledged mutations and retry outcomes survive warm audit, joined drain and cold restore | +| Stability | Bounded memory, native-job occupancy and debt; debt has no sustained positive slope | + +Qualify read-only, write-only and simultaneous 10K-write/50K-read load separately. +Count the client, provider and followers separately from the serving node. +Docker can simulate the deployment, but sharing one 8-CPU VM between all roles +does not qualify an 8-CPU serving node. Report KV overwrite and SQL commands with +the durable request/result ledger separately. tmpfs results do not qualify +physical-media durability. Preserve the stronger historical comparison gates in +the [workspace proposal](../../../docs/write-performance-proposal.md); this node +capacity contract does not turn a failed earlier profile into a passing one. + +## Canonical write path + +```mermaid +flowchart LR + A[Bounded admission before SQL] --> B[Mutation and retry result commit together] + B --> C[Exact complete capture assignment] + C --> D[Ordered shared native log] + D --> E[Authenticated follower durable proof] + E --> F[Fleet ACK] + C --> G[One admitted cross-Cell bundle] + G --> H[Original bindings and exact range verification] + H --> I[Canonical node selector CAS] + I --> J[Bucket ACK and proven visibility] + I --> K[Bounded joined root materializer] + K --> L[Exact root cohort checkpoint] +``` + +Uploaded bytes are not authority. Bundle ACKs remain disabled until the actor, +worker, read/retry endpoint, recovery and lifecycle consumers agree on one +opaque complete proof. Do not advance the follower reclamation frontier before +failed-owner recovery understands the selected bundle prefix. + +## Locator density and checkpoint cost + +Under the measured small-tail four-PUT materialization path, 64 commands per +shared selection and 64 roots per checkpoint give the conditional cost +`2/64 + (4 + 2/64)/K`. At most 0.05 publication PUTs/command therefore requires +at least **215 commands per Cell checkpoint**, before compaction, retries and +collection. Larger tails can leave the four-PUT path and must be measured. + +The prior 32 inline frame locators cannot represent this density under uniform +traffic. Increasing only that array also increases every sibling's shard read. +The implemented development representation separates small binding rows from authenticated, +independently addressed locator histories: + +- Keep the 256-shard fixed authenticated header and original binding pins. +- Keep unchanged binding shards and histories by exact object/range/digest. +- Store a Cell history in a bounded extent in the same selection object; adding + a history adds bytes, not a separate PUT. Never rewrite old native bodies. +- Load histories only for the requested Cell/cohort, retaining sibling history + references without decoding their contents. Preflight aggregate selected + shard and history metadata against the existing 4 MiB protocol-input bound. +- Bound each history to 256 exact frame references and 32 KiB encoded metadata; + preserve the 4 MiB verified native-suffix byte bound and 64 frames/selection. +- New selectors use a distinct format version. Existing complete catalogs and + inline indexed shards remain readable; upgrade every recovery consumer before + selecting the new format. Use a fresh development prefix during rollout. + +Those are protocol bounds, not host admission. The materializer scheduler must +charge retained history, native verification, scratch and outcomes to the node +ledgers. Uniform 10K writes over 2,000 Cells means about five commands/Cell/s: +215 commands span about 43 seconds. Measure the actual native bytes retained +over that interval and reject the model if the 16-GiB node cannot hold its debt. + +## Remaining implementation and exit gates + +| Work | Required verification | +| --- | --- | +| Dense histories | 215 exact commands without checkpoint; corrupt/missing/cross-scope history rejection; bounded selected metadata; byte-identical cold restore; measured larger-tail cost | +| Actor ACK/read/retry integration | Response and visible retry/query state use the same original proof; captured ranges release only after exact matching; unproven suffix remains hidden | +| Canonical coverage frontier | Bundle selector and contiguous native coverage selected atomically; no second authority CAS; ambiguous replies, heartbeat, lease loss and cancellation tested | +| Admitted materializer | One bounded node scheduler, fair cohorts, debt limit and joined native/storage jobs on cancellation and shutdown | +| Failed-owner recovery | Seal complete issued suffix including prior Fleet ACKs; selected prefix plus follower tail restore exactly before closing all bindings or admitting replacement writers | +| Safe collection | Complete cross-Cell root/catalog/proof/reference inventory, admitted scans, quiesced writes and grace-qualified deletion | +| Read efficiency | Bounded workers and views, proven snapshots, minimal per-query authority I/O; overload retains renewal/publication headroom | +| Paired qualification | Node capacity contract and unchanged durability/read/drain gates pass; otherwise report measured gaps with source and binary identities | + +Implementation and prior measured results are tracked in the +[bundle delivery record](../../../docs/bundle-coverage-implementation.md) and +[WAL comparison](../../../docs/pr67-normal-wal-reevaluation.md). Keep bulk logs, +raw request ledgers and source manifests outside the repository. diff --git a/crates/cellule-runtime/src/node/bundle/binding.rs b/crates/cellule-runtime/src/node/bundle/binding.rs index 9dcd4e9a..edcc3fea 100644 --- a/crates/cellule-runtime/src/node/bundle/binding.rs +++ b/crates/cellule-runtime/src/node/bundle/binding.rs @@ -52,14 +52,11 @@ impl NodeDirectory { .bundle .ok_or(Error::Node("bundle lane is absent"))?; let value = control.value(); - let shards = [index::shard( - authority.layout().application_id(), - value.cell.as_bytes(), - )] - .into_iter() - .collect(); + let cells = [(*authority.layout().application_id(), *value.cell.as_bytes())] + .into_iter() + .collect(); let mut catalog = - store::load_catalog_shards(&self.layout, observed.advertisement.session, head, &shards) + store::load_catalog_cells(&self.layout, observed.advertisement.session, head, &cells) .await?; if value.state != ControlState::Serving || value.recovery.is_some() diff --git a/crates/cellule-runtime/src/node/bundle/closure.rs b/crates/cellule-runtime/src/node/bundle/closure.rs index 232fd778..2fcfb408 100644 --- a/crates/cellule-runtime/src/node/bundle/closure.rs +++ b/crates/cellule-runtime/src/node/bundle/closure.rs @@ -41,7 +41,7 @@ impl NodeDirectory { .bundle .ok_or(Error::Node("bundle lane is absent"))?; let mut pins = std::collections::HashSet::new(); - let mut shards = std::collections::BTreeSet::new(); + let mut cells = std::collections::BTreeSet::new(); for (_, proof) in checkpoints { if proof.pin.session != observed.advertisement.session || proof.pin.epoch != head.epoch @@ -49,13 +49,13 @@ impl NodeDirectory { { return Err(Error::Fenced); } - shards.insert(index::shard( - proof.binding.application.as_bytes(), - proof.binding.control.cell.as_bytes(), + cells.insert(( + *proof.binding.application.as_bytes(), + *proof.binding.control.cell.as_bytes(), )); } let mut catalog = - store::load_catalog_shards(&self.layout, observed.advertisement.session, head, &shards) + store::load_catalog_cells(&self.layout, observed.advertisement.session, head, &cells) .await?; let mut changed = false; for (authority, proof) in checkpoints { @@ -118,14 +118,11 @@ impl NodeDirectory { .bundle .ok_or(Error::Node("bundle lane is absent"))?; let scope = issued.scope(); - let shards = [index::shard( - scope.application.as_bytes(), - scope.cell.as_bytes(), - )] - .into_iter() - .collect(); + let cells = [(*scope.application.as_bytes(), *scope.cell.as_bytes())] + .into_iter() + .collect(); let mut catalog = - store::load_catalog_shards(&self.layout, pin.session, head, &shards).await?; + store::load_catalog_cells(&self.layout, pin.session, head, &cells).await?; if pin.epoch != catalog.epoch { return Err(Error::Fenced); } @@ -169,14 +166,11 @@ impl NodeDirectory { .bundle .ok_or(Error::Node("bundle lane is absent"))?; let scope = issued.scope(); - let shards = [index::shard( - scope.application.as_bytes(), - scope.cell.as_bytes(), - )] - .into_iter() - .collect(); + let cells = [(*scope.application.as_bytes(), *scope.cell.as_bytes())] + .into_iter() + .collect(); let mut catalog = - store::load_catalog_shards(&self.layout, pin.session, head, &shards).await?; + store::load_catalog_cells(&self.layout, pin.session, head, &cells).await?; let binding = catalog.binding_mut(pin.digest)?; if pin.epoch != head.epoch || issued.leader_session() != pin.session diff --git a/crates/cellule-runtime/src/node/bundle/codec.rs b/crates/cellule-runtime/src/node/bundle/codec.rs index 86248ac4..5e6bed4f 100644 --- a/crates/cellule-runtime/src/node/bundle/codec.rs +++ b/crates/cellule-runtime/src/node/bundle/codec.rs @@ -29,6 +29,9 @@ fn metadata(catalog: &Catalog, frames: usize) -> Result { e.write_u64(catalog.selected_through)?; e.write_count(catalog.bindings.len())?; for binding in &catalog.bindings { + if binding.locators.len() > MAX_INLINE_LOCATORS { + return Err(Error::Capacity("inline bundle locator count")); + } e.write_bytes(binding.application.as_bytes())?; e.write_u64(binding.first_commit)?; e.write_bytes(&binding.control.encode()?)?; @@ -156,7 +159,7 @@ fn decode_inner(body: &Bytes, complete_object: bool) -> Result { let selected_commit = d.read_u64()?; let selected_position = position_read(&mut d)?; let count = d.read_count()?; - if count > MAX_LOCATORS { + if count > MAX_INLINE_LOCATORS { return Err(Error::Capacity("bundle locator count")); } let mut locators = Vec::with_capacity(count); diff --git a/crates/cellule-runtime/src/node/bundle/index/codec.rs b/crates/cellule-runtime/src/node/bundle/index/codec.rs index 8218475e..89fee27f 100644 --- a/crates/cellule-runtime/src/node/bundle/index/codec.rs +++ b/crates/cellule-runtime/src/node/bundle/index/codec.rs @@ -2,7 +2,7 @@ use super::*; use crate::codec::{BoundedDecoder, BoundedEncoder, read_fixed}; -fn write_extent(e: &mut BoundedEncoder, extent: &Locator) -> Result<()> { +pub(super) fn write_extent(e: &mut BoundedEncoder, extent: &Locator) -> Result<()> { e.write_bool(extent.object.is_some())?; if let Some(object) = extent.object { e.write_bytes(object.as_bytes())?; @@ -12,7 +12,7 @@ fn write_extent(e: &mut BoundedEncoder, extent: &Locator) -> Result<()> { e.write_bytes(extent.frame_digest.as_bytes())?; Ok(()) } -fn read_extent(d: &mut BoundedDecoder<'_>) -> Result { +pub(super) fn read_extent(d: &mut BoundedDecoder<'_>) -> Result { let object = if d.read_bool()? { Some(Digest::from_bytes(read_fixed( d, @@ -53,7 +53,7 @@ pub(super) fn encode(root: &Root) -> Result { } let payload = e.finish(); let mut header = Vec::with_capacity(HEADER_BYTES); - header.extend_from_slice(MAGIC); + header.extend_from_slice(if root.detached { DENSE_MAGIC } else { MAGIC }); header.extend_from_slice(&(payload.len() as u32).to_be_bytes()); header.extend_from_slice(&payload); header.resize(HEADER_BYTES, 0); @@ -61,7 +61,9 @@ pub(super) fn encode(root: &Root) -> Result { } pub(super) fn decode(header: &[u8]) -> Result { - if header.len() != HEADER_BYTES || header.get(..8) != Some(MAGIC.as_slice()) { + if header.len() != HEADER_BYTES + || !matches!(header.get(..8), Some(magic) if magic == MAGIC || magic == DENSE_MAGIC) + { return Err(Error::Node("unsupported bundle index header")); } let length = u32::from_be_bytes( @@ -110,6 +112,7 @@ pub(super) fn decode(header: &[u8]) -> Result { } d.finish()?; let root = Root { + detached: header.get(..8) == Some(DENSE_MAGIC.as_slice()), session, epoch, predecessor, @@ -156,12 +159,12 @@ fn validate(root: &Root) -> Result<()> { } local_end += frame.bytes; } - if local_end != root.object_bytes { + if local_end > root.object_bytes || (!root.detached && local_end != root.object_bytes) { return Err(Error::Node("bundle index object length differs")); } Ok(()) } -fn validate_extent(extent: &Locator) -> Result<()> { +pub(super) fn validate_extent(extent: &Locator) -> Result<()> { if extent.offset < HEADER_BYTES as u64 || extent.bytes == 0 || extent diff --git a/crates/cellule-runtime/src/node/bundle/index/dense.rs b/crates/cellule-runtime/src/node/bundle/index/dense.rs new file mode 100644 index 00000000..5af72df5 --- /dev/null +++ b/crates/cellule-runtime/src/node/bundle/index/dense.rs @@ -0,0 +1,276 @@ +//! One immutable upload holds changed shards, native frames and small histories. +use super::*; + +pub(super) fn encode( + catalog: &mut Catalog, + frames: &[cellule_ltx::VerifiedNodeFrame], +) -> Result<(Bytes, Digest)> { + catalog.validate()?; + if frames.len() > MAX_FRAMES { + return Err(Error::Capacity("bundle frame count")); + } + let groups = grouped(catalog); + let mut shards = match &catalog.index { + Some(index) => { + if index.root.session != catalog.session || index.root.epoch != catalog.epoch { + return Err(Error::Fenced); + } + if groups.keys().any(|id| !index.loaded.contains_key(id)) { + return Err(Error::Node("bundle modifies an unloaded catalog shard")); + } + index.root.shards.clone() + } + None => vec![None; SHARDS], + }; + let changed: BTreeSet<_> = match &catalog.index { + Some(index) => index + .loaded + .iter() + .filter_map(|(id, original)| { + (groups.get(id).map(Vec::as_slice).unwrap_or(&[]) != original.as_slice()) + .then_some(*id) + }) + .collect(), + None => groups.keys().copied().collect(), + }; + let mut histories = BTreeMap::new(); + let mut new = BTreeSet::new(); + for binding in &catalog.bindings { + if !changed.contains(&binding_shard(binding)) { + continue; + } + let pin = history::pin(binding)?; + if let Some(original) = catalog + .index + .as_ref() + .and_then(|index| index.histories.get(&pin)) + { + let same = match &original.loaded { + Some(loaded) => loaded == &binding.locators, + None => binding.locators.as_slice() == std::slice::from_ref(&original.extent), + }; + if same { + histories.insert(pin, original.clone()); + continue; + } + if original.loaded.is_none() { + return Err(Error::Node("bundle modifies an unloaded Cell history")); + } + } + if binding.locators.is_empty() { + continue; + } + let bytes = history::encode(catalog.session, catalog.epoch, binding)?; + let native_bytes = binding + .locators + .iter() + .try_fold(0_u64, |total, locator| total.checked_add(locator.bytes)) + .filter(|bytes| *bytes <= MAX_SUFFIX_BYTES) + .ok_or(Error::Capacity("bundle history native bytes"))?; + histories.insert( + pin, + history::History { + extent: Locator { + object: None, + offset: 0, + bytes: bytes.len() as u64, + frame_digest: Digest::from_bytes([0; 32]), + }, + count: binding.locators.len(), + native_bytes, + loaded: Some(binding.locators.clone()), + }, + ); + new.insert(pin); + } + let mut offset = HEADER_BYTES as u64; + for id in &changed { + let rows = groups.get(id).cloned().unwrap_or_default(); + if rows.is_empty() { + shards[usize::from(*id)] = None; + continue; + } + let bytes = history::encode_leaf(&leaf(catalog, rows), &histories)?.len() as u64; + shards[usize::from(*id)] = Some(Shard { + extent: Locator { + object: None, + offset, + bytes, + frame_digest: Digest::from_bytes([0; 32]), + }, + bindings: groups[id].len(), + }); + offset = offset + .checked_add(bytes) + .ok_or(Error::Capacity("bundle shard bytes"))?; + } + let mut native = Vec::with_capacity(frames.len()); + for frame in frames { + let digest = Digest::from_bytes(frame.digest()); + let mut matches = 0; + for binding in &mut catalog.bindings { + let id = binding_shard(binding); + for locator in &mut binding.locators { + if locator.object.is_none() && locator.frame_digest == digest { + if !changed.contains(&id) { + return Err(Error::Node("native frame belongs to an unchanged shard")); + } + locator.offset = offset; + locator.bytes = frame.encoded().len() as u64; + matches += 1; + } + } + } + if matches != 1 { + return Err(Error::Node("bundle frame locator is not unique")); + } + native.push(Locator { + object: None, + offset, + bytes: frame.encoded().len() as u64, + frame_digest: digest, + }); + offset = offset + .checked_add(frame.encoded().len() as u64) + .ok_or(Error::Capacity("bundle bytes"))?; + } + let mut history_bodies = Vec::new(); + for pin in new { + let binding = catalog + .bindings + .iter() + .find(|binding| history::pin(binding).ok() == Some(pin)) + .ok_or(Error::Node("bundle history binding is absent"))?; + let bytes = history::encode(catalog.session, catalog.epoch, binding)?; + let reference = histories + .get_mut(&pin) + .ok_or(Error::Node("bundle history plan is absent"))?; + if bytes.len() as u64 != reference.extent.bytes { + return Err(Error::Node("bundle history size changed during encoding")); + } + reference.extent.offset = offset; + reference.extent.frame_digest = Digest::from_bytes(*blake3::hash(&bytes).as_bytes()); + reference.loaded = Some(binding.locators.clone()); + offset = offset + .checked_add(bytes.len() as u64) + .ok_or(Error::Capacity("bundle bytes"))?; + history_bodies.push(bytes); + } + if offset > MAX_BUNDLE_BYTES { + return Err(Error::Capacity("bundle bytes")); + } + let groups = grouped(catalog); + let mut bodies = Vec::new(); + for id in &changed { + if let Some(reference) = &mut shards[usize::from(*id)] { + let body = history::encode_leaf(&leaf(catalog, groups[id].clone()), &histories)?; + if body.len() as u64 != reference.extent.bytes { + return Err(Error::Node("bundle shard size changed during encoding")); + } + reference.extent.frame_digest = Digest::from_bytes(*blake3::hash(&body).as_bytes()); + bodies.push(body); + } + } + let root = Root { + detached: true, + session: catalog.session, + epoch: catalog.epoch, + predecessor: catalog.predecessor, + selected_through: catalog.selected_through, + object_bytes: offset, + shards, + frames: native, + }; + let header = codec::encode(&root)?; + let digest = Digest::from_bytes(*blake3::hash(&header).as_bytes()); + let mut body = Vec::with_capacity(offset as usize); + body.extend_from_slice(&header); + for shard in bodies { + body.extend_from_slice(&shard); + } + for frame in frames { + body.extend_from_slice(frame.encoded()); + } + for history in history_bodies { + body.extend_from_slice(&history); + } + let body = Bytes::from(body); + verify_local(&body, &root)?; + Ok((body, digest)) +} + +fn verify_local(body: &Bytes, expected: &Root) -> Result<()> { + let root = codec::decode(&body[..HEADER_BYTES])?; + if &root != expected || body.len() as u64 != root.object_bytes { + return Err(Error::Node("bundle index self-verification differs")); + } + let mut offset = HEADER_BYTES as u64; + let mut local_histories = BTreeMap::new(); + for (id, shard) in root.shards.iter().enumerate() { + let Some(shard) = shard else { + continue; + }; + if shard.extent.object.is_some() { + continue; + } + if shard.extent.offset != offset { + return Err(Error::Node("bundle local shard extents are not canonical")); + } + let (leaf, histories) = + decode_leaf_with_histories(extent_bytes(body, &shard.extent)?, shard, id as u8, &root)?; + for binding in leaf.bindings { + let pin = history::pin(&binding)?; + if let Some(history) = histories + .get(&pin) + .filter(|history| history.extent.object.is_none()) + && local_histories + .insert(pin, (binding, history.clone())) + .is_some() + { + return Err(Error::Node("bundle local history is not unique")); + } + } + offset += shard.extent.bytes; + } + for frame in &root.frames { + if frame.offset != offset { + return Err(Error::Node("bundle local native extents differ")); + } + extent_bytes(body, frame)?; + offset += frame.bytes; + } + let mut native_locators = Vec::new(); + for (binding, history) in local_histories.values() { + if history.extent.offset != offset { + return Err(Error::Node("bundle local histories are not canonical")); + } + let locators = history::decode( + &extent_bytes(body, &history.extent)?, + root.session, + root.epoch, + binding, + history, + )?; + native_locators.extend( + locators + .into_iter() + .filter(|locator| locator.object.is_none()), + ); + offset += history.extent.bytes; + } + if offset != root.object_bytes + || native_locators.len() != root.frames.len() + || root.frames.iter().any(|frame| { + native_locators + .iter() + .filter(|locator| *locator == frame) + .count() + != 1 + }) + { + return Err(Error::Node( + "bundle local history/native extent coverage differs", + )); + } + Ok(()) +} diff --git a/crates/cellule-runtime/src/node/bundle/index/history.rs b/crates/cellule-runtime/src/node/bundle/index/history.rs new file mode 100644 index 00000000..16c7d6b9 --- /dev/null +++ b/crates/cellule-runtime/src/node/bundle/index/history.rs @@ -0,0 +1,201 @@ +//! Detached authenticated histories: sibling rows retain references, not arrays. +use super::*; +use crate::codec::{BoundedDecoder, BoundedEncoder, read_fixed}; + +pub(super) const MAX_HISTORY_BYTES: u64 = 32 << 10; +const LEAF_MAGIC: &[u8] = b"CBL3"; +const HISTORY_MAGIC: &[u8] = b"CLH3"; + +#[derive(Clone, Debug, PartialEq, Eq)] +pub(super) struct History { + pub(super) extent: Locator, + pub(super) count: usize, + pub(super) native_bytes: u64, + // None means the binding carries exactly the authenticated extent above. + // Such a row cannot be verified as native frames or modified as a suffix. + pub(super) loaded: Option>, +} + +pub(super) fn pin(binding: &Binding) -> Result<[u8; 32]> { + binding + .control + .bundle_binding + .map(|pin| *pin.digest.as_bytes()) + .ok_or(Error::Node("bundle history lacks Cell pin")) +} + +pub(super) fn encode(session: SessionId, epoch: u64, binding: &Binding) -> Result { + if binding.locators.is_empty() || binding.locators.len() > MAX_LOCATORS { + return Err(Error::Capacity("bundle history count")); + } + let mut e = BoundedEncoder::new(MAX_HISTORY_BYTES as u32)?; + e.write_bytes(HISTORY_MAGIC)?; + e.write_bytes(session.as_bytes())?; + e.write_u64(epoch)?; + e.write_bytes(&pin(binding)?)?; + e.write_count(binding.locators.len())?; + for locator in &binding.locators { + super::codec::write_extent(&mut e, locator)?; + } + Ok(Bytes::from(e.finish())) +} + +pub(super) fn decode( + body: &Bytes, + session: SessionId, + epoch: u64, + binding: &Binding, + history: &History, +) -> Result> { + if body.len() as u64 != history.extent.bytes + || *blake3::hash(body).as_bytes() != *history.extent.frame_digest.as_bytes() + { + return Err(Error::Node("bundle history digest differs")); + } + let mut d = BoundedDecoder::new(body, MAX_HISTORY_BYTES as u32)?; + if d.read_bytes()? != HISTORY_MAGIC + || read_fixed::<16>(&mut d, "bundle history session length")? != *session.as_bytes() + || d.read_u64()? != epoch + || read_fixed::<32>(&mut d, "bundle history pin length")? != pin(binding)? + { + return Err(Error::Fenced); + } + let count = d.read_count()?; + if count == 0 || count > MAX_LOCATORS || count != history.count { + return Err(Error::Capacity("bundle history count")); + } + let mut locators = Vec::with_capacity(count); + let mut native_bytes = 0_u64; + for _ in 0..count { + let locator = super::codec::read_extent(&mut d)?; + // Native locators from the complete legacy format may precede the + // fixed index header, so the history preserves that older byte offset. + if locator.bytes == 0 + || locator + .offset + .checked_add(locator.bytes) + .is_none_or(|end| end > MAX_BUNDLE_BYTES) + || locator + .frame_digest + .as_bytes() + .iter() + .all(|byte| *byte == 0) + || locator + .object + .is_some_and(|object| object.as_bytes().iter().all(|byte| *byte == 0)) + { + return Err(Error::Node("invalid bundle history native extent")); + } + native_bytes = native_bytes + .checked_add(locator.bytes) + .filter(|bytes| *bytes <= MAX_SUFFIX_BYTES) + .ok_or(Error::Capacity("bundle history native bytes"))?; + locators.push(locator); + } + d.finish()?; + if native_bytes != history.native_bytes { + return Err(Error::Node("bundle history native byte count differs")); + } + Ok(locators) +} + +pub(super) fn encode_leaf( + catalog: &Catalog, + histories: &BTreeMap<[u8; 32], History>, +) -> Result { + let mut compact = catalog.clone(); + compact.index = None; + for binding in &mut compact.bindings { + if let Some(history) = histories.get(&pin(binding)?) { + binding.locators = vec![history.extent.clone()]; + } else if !binding.locators.is_empty() { + return Err(Error::Node("bundle leaf lacks detached history")); + } + } + let inline = super::super::codec::encode_leaf(&compact)?; + let mut e = BoundedEncoder::new(MAX_BUNDLE_BYTES as u32)?; + e.write_bytes(LEAF_MAGIC)?; + e.write_bytes(&inline)?; + e.write_count( + compact + .bindings + .iter() + .filter(|binding| !binding.locators.is_empty()) + .count(), + )?; + for binding in &compact.bindings { + if let Some(history) = histories.get(&pin(binding)?) { + e.write_bytes(&pin(binding)?)?; + e.write_count(history.count)?; + e.write_u64(history.native_bytes)?; + } + } + Ok(Bytes::from(e.finish())) +} + +pub(super) fn decode_leaf(body: &Bytes) -> Result<(Catalog, BTreeMap<[u8; 32], History>)> { + if body.get(..8) != Some(b"\0\0\0\x04CBL3".as_slice()) { + return Ok((super::super::codec::decode_leaf(body)?, BTreeMap::new())); + } + let mut d = BoundedDecoder::new(body, MAX_BUNDLE_BYTES as u32)?; + if d.read_bytes()? != LEAF_MAGIC { + return Err(Error::Node("unsupported detached bundle leaf")); + } + let catalog = super::super::codec::decode_leaf(&Bytes::copy_from_slice(d.read_bytes()?))?; + let count = d.read_count()?; + if count + != catalog + .bindings + .iter() + .filter(|binding| !binding.locators.is_empty()) + .count() + { + return Err(Error::Node("bundle history descriptor count differs")); + } + let mut histories = BTreeMap::new(); + for binding in &catalog.bindings { + if binding.locators.is_empty() { + continue; + } + if binding.locators.len() != 1 + || read_fixed::<32>(&mut d, "bundle history pin length")? != pin(binding)? + { + return Err(Error::Node("bundle history descriptor scope differs")); + } + let history = History { + extent: binding.locators[0].clone(), + count: d.read_count()?, + native_bytes: d.read_u64()?, + loaded: None, + }; + super::codec::validate_extent(&history.extent)?; + if history.extent.bytes > MAX_HISTORY_BYTES + || history.count == 0 + || history.count > MAX_LOCATORS + || history.native_bytes == 0 + || history.native_bytes > MAX_SUFFIX_BYTES + || histories.insert(pin(binding)?, history).is_some() + { + return Err(Error::Capacity("bundle history descriptor bounds")); + } + } + d.finish()?; + Ok((catalog, histories)) +} + +pub(in crate::node::bundle) fn validate_deferred( + catalog: &Catalog, + binding: &Binding, +) -> Result<()> { + let pin = pin(binding)?; + if let Some(history) = catalog + .index + .as_ref() + .and_then(|index| index.histories.get(&pin)) + && history.loaded.is_none() + && binding.locators.as_slice() != std::slice::from_ref(&history.extent) + { + return Err(Error::Node("bundle modifies an unloaded Cell history")); + } + Ok(()) +} diff --git a/crates/cellule-runtime/src/node/bundle/index/io.rs b/crates/cellule-runtime/src/node/bundle/index/io.rs index 3a74dc32..9f9fab66 100644 --- a/crates/cellule-runtime/src/node/bundle/index/io.rs +++ b/crates/cellule-runtime/src/node/bundle/index/io.rs @@ -12,7 +12,7 @@ async fn load_root( .store() .range_get(&path, 0..HEADER_BYTES as u64) .await?; - if header.get(..8) != Some(MAGIC.as_slice()) { + if !matches!(header.get(..8), Some(magic) if magic == MAGIC || magic == DENSE_MAGIC) { return Ok(None); } if *blake3::hash(&header).as_bytes() != *head.digest.as_bytes() { @@ -38,7 +38,7 @@ async fn load_rows( root: &Root, shard: &Shard, id: u8, -) -> Result> { +) -> Result<(Vec, BTreeMap<[u8; 32], history::History>)> { let extent = &shard.extent; let object = extent .object @@ -54,7 +54,7 @@ async fn load_rows( extent.offset..extent.offset + extent.bytes, ) .await?; - let mut leaf = decode_leaf(bytes, shard, id, root)?; + let (mut leaf, mut histories) = decode_leaf_with_histories(bytes, shard, id, root)?; for locator in leaf .bindings .iter_mut() @@ -64,14 +64,43 @@ async fn load_rows( locator.object = Some(object); } } - Ok(leaf.bindings) + for history in histories.values_mut() { + if history.extent.object.is_none() { + history.extent.object = Some(object); + } + } + Ok((leaf.bindings, histories)) } +#[cfg(test)] pub(in crate::node::bundle) async fn load( layout: &cellule_ltx::CellStorageLayout, session: SessionId, head: NodeBundleHead, wanted: Option<&BTreeSet>, +) -> Result { + load_inner(layout, session, head, wanted, None).await +} + +pub(in crate::node::bundle) async fn load_cells( + layout: &cellule_ltx::CellStorageLayout, + session: SessionId, + head: NodeBundleHead, + cells: &BTreeSet, +) -> Result { + let shards = cells + .iter() + .map(|(application, cell)| shard(application, cell)) + .collect(); + load_inner(layout, session, head, Some(&shards), Some(cells)).await +} + +async fn load_inner( + layout: &cellule_ltx::CellStorageLayout, + session: SessionId, + head: NodeBundleHead, + wanted: Option<&BTreeSet>, + cells: Option<&BTreeSet>, ) -> Result { let Some(root) = load_root(layout, session, head).await? else { return super::super::store::load_legacy_catalog(layout, session, head).await; @@ -91,15 +120,21 @@ pub(in crate::node::bundle) async fn load( } let mut loaded = BTreeMap::new(); let mut bindings = Vec::new(); + let mut histories = BTreeMap::new(); for id in 0..SHARDS { let id = id as u8; if wanted.is_some_and(|wanted| !wanted.contains(&id)) { continue; } - let rows = match &root.shards[usize::from(id)] { + let (rows, leaf_histories) = match &root.shards[usize::from(id)] { Some(shard) => load_rows(layout, &root, shard, id).await?, - None => Vec::new(), + None => (Vec::new(), BTreeMap::new()), }; + for (pin, history) in leaf_histories { + if histories.insert(pin, history).is_some() { + return Err(Error::Node("bundle inventory repeats a Cell pin")); + } + } bindings.extend(rows.iter().cloned()); loaded.insert(id, rows); } @@ -109,13 +144,84 @@ pub(in crate::node::bundle) async fn load( .bundle_binding .map(|pin| *pin.digest.as_bytes()) }); + // The leaf authenticates the size of each separately addressed history. + // Reject the aggregate before the first history read, and leave unrelated + // sibling histories as authenticated references for copy-on-write updates. + for binding in &bindings { + if cells.is_some_and(|cells| { + !cells.contains(&( + *binding.application.as_bytes(), + *binding.control.cell.as_bytes(), + )) + }) { + continue; + } + if let Some(history) = histories.get(&history::pin(binding)?) { + selected_bytes = selected_bytes + .checked_add(history.extent.bytes) + .filter(|bytes| *bytes <= MAX_BUNDLE_BYTES) + .ok_or(Error::Capacity("bundle selected history bytes"))?; + } + } + for binding in &mut bindings { + if cells.is_some_and(|cells| { + !cells.contains(&( + *binding.application.as_bytes(), + *binding.control.cell.as_bytes(), + )) + }) { + continue; + } + let Some(history) = histories.get_mut(&history::pin(binding)?) else { + continue; + }; + let object = history + .extent + .object + .ok_or(Error::Node("unresolved bundle history"))?; + let bytes = layout + .store() + .range_get( + &layout.node_coverage_bundle_path( + session.as_bytes(), + head.epoch, + object.as_bytes(), + ), + history.extent.offset..history.extent.offset + history.extent.bytes, + ) + .await?; + let mut locators = history::decode(&bytes, session, head.epoch, binding, history)?; + for locator in &mut locators { + if locator.object.is_none() { + locator.object = Some(object); + } + } + history.loaded = Some(locators.clone()); + binding.locators = locators; + } + // Snapshot after requested histories have loaded: hydration itself must + // not rewrite a shard that the caller never changes. + let mut hydrated = BTreeMap::>::new(); + for binding in &bindings { + hydrated + .entry(binding_shard(binding)) + .or_default() + .push(binding.clone()); + } + for (id, rows) in &mut loaded { + *rows = hydrated.remove(id).unwrap_or_default(); + } let catalog = Catalog { session, epoch: head.epoch, predecessor: root.predecessor, selected_through: head.selected_through, bindings, - index: Some(LoadedIndex { root, loaded }), + index: Some(LoadedIndex { + root, + loaded, + histories, + }), }; catalog.validate()?; Ok(catalog) @@ -136,7 +242,7 @@ pub(in crate::node::bundle) async fn ensure_drained( let mut pins = std::collections::HashSet::new(); for (id, shard) in root.shards.iter().enumerate() { let Some(shard) = shard else { continue }; - let rows = load_rows(layout, &root, shard, id as u8).await?; + let (rows, _) = load_rows(layout, &root, shard, id as u8).await?; for binding in &rows { let pin = binding .control diff --git a/crates/cellule-runtime/src/node/bundle/index/mod.rs b/crates/cellule-runtime/src/node/bundle/index/mod.rs index 648d05cf..53bc48dc 100644 --- a/crates/cellule-runtime/src/node/bundle/index/mod.rs +++ b/crates/cellule-runtime/src/node/bundle/index/mod.rs @@ -7,14 +7,33 @@ use super::*; use std::collections::{BTreeMap, BTreeSet}; mod codec; +mod dense; +mod history; mod io; #[cfg(test)] mod tests; -pub(super) use io::{ensure_drained, load}; +pub(super) use history::validate_deferred; +#[cfg(test)] +pub(super) use io::load; +#[cfg(test)] +pub(super) use tests::encode_inline; + +#[cfg(test)] +pub(super) fn history_extent(catalog: &Catalog, pin: Digest) -> Option { + catalog + .index + .as_ref()? + .histories + .get(pin.as_bytes()) + .map(|history| history.extent.clone()) +} +pub(super) use io::{ensure_drained, load_cells}; pub(super) const HEADER_BYTES: usize = 32 << 10; const SHARDS: usize = 256; const MAGIC: &[u8; 8] = b"\0\0\0\x04CNB2"; +const DENSE_MAGIC: &[u8; 8] = b"\0\0\0\x04CNB3"; +pub(super) type CellKey = ([u8; 16], [u8; 32]); #[derive(Clone, Debug, PartialEq, Eq)] struct Shard { @@ -24,6 +43,7 @@ struct Shard { #[derive(Clone, Debug, PartialEq, Eq)] struct Root { + detached: bool, session: SessionId, epoch: u64, predecessor: Option, @@ -37,6 +57,7 @@ struct Root { pub(super) struct LoadedIndex { root: Root, loaded: BTreeMap>, + histories: BTreeMap<[u8; 32], history::History>, } /// The shard key includes the application, so equal Cell IDs in applications @@ -82,170 +103,7 @@ pub(super) fn encode( catalog: &mut Catalog, frames: &[cellule_ltx::VerifiedNodeFrame], ) -> Result<(Bytes, Digest)> { - catalog.validate()?; - if frames.len() > MAX_FRAMES { - return Err(Error::Capacity("bundle frame count")); - } - let groups = grouped(catalog); - let mut shards = match &catalog.index { - Some(index) => { - if index.root.session != catalog.session || index.root.epoch != catalog.epoch { - return Err(Error::Fenced); - } - // A partial catalog may modify only a shard it authenticated first. - if groups.keys().any(|id| !index.loaded.contains_key(id)) { - return Err(Error::Node("bundle modifies an unloaded catalog shard")); - } - index.root.shards.clone() - } - None => vec![None; SHARDS], - }; - let changed: BTreeSet = match &catalog.index { - Some(index) => index - .loaded - .iter() - .filter_map(|(id, original)| { - (groups.get(id).map(Vec::as_slice).unwrap_or(&[]) != original.as_slice()) - .then_some(*id) - }) - .collect(), - None => groups.keys().copied().collect(), - }; - // Native extents follow the changed shards; changing numeric offsets never - // changes the leaf encoding length. Compute sizes before filling offsets. - let mut offset = HEADER_BYTES as u64; - for id in &changed { - let rows = groups.get(id).cloned().unwrap_or_default(); - if rows.is_empty() { - shards[usize::from(*id)] = None; - } else { - let bytes = super::codec::encode_leaf(&leaf(catalog, rows))?.len() as u64; - shards[usize::from(*id)] = Some(Shard { - extent: Locator { - object: None, - offset, - bytes, - frame_digest: Digest::from_bytes([0; 32]), - }, - bindings: groups[id].len(), - }); - offset = offset - .checked_add(bytes) - .ok_or(Error::Capacity("bundle shard bytes"))?; - } - } - let mut native = Vec::with_capacity(frames.len()); - for frame in frames { - let digest = Digest::from_bytes(frame.digest()); - let mut matches = 0; - for binding in &mut catalog.bindings { - let id = binding_shard(binding); - for locator in &mut binding.locators { - if locator.object.is_none() && locator.frame_digest == digest { - if !changed.contains(&id) { - return Err(Error::Node("native frame belongs to an unchanged shard")); - } - locator.offset = offset; - locator.bytes = frame.encoded().len() as u64; - matches += 1; - } - } - } - if matches != 1 { - return Err(Error::Node("bundle frame locator is not unique")); - } - native.push(Locator { - object: None, - offset, - bytes: frame.encoded().len() as u64, - frame_digest: digest, - }); - offset = offset - .checked_add(frame.encoded().len() as u64) - .ok_or(Error::Capacity("bundle bytes"))?; - } - if offset > MAX_BUNDLE_BYTES { - return Err(Error::Capacity("bundle bytes")); - } - let groups = grouped(catalog); - let mut bodies = Vec::new(); - for id in &changed { - if let Some(reference) = &mut shards[usize::from(*id)] { - let body = super::codec::encode_leaf(&leaf(catalog, groups[id].clone()))?; - if body.len() as u64 != reference.extent.bytes { - return Err(Error::Node("bundle shard size changed during encoding")); - } - reference.extent.frame_digest = Digest::from_bytes(*blake3::hash(&body).as_bytes()); - bodies.push(body); - } - } - let root = Root { - session: catalog.session, - epoch: catalog.epoch, - predecessor: catalog.predecessor, - selected_through: catalog.selected_through, - object_bytes: offset, - shards, - frames: native, - }; - let header = codec::encode(&root)?; - let digest = Digest::from_bytes(*blake3::hash(&header).as_bytes()); - let mut body = Vec::with_capacity(offset as usize); - body.extend_from_slice(&header); - for leaf in bodies { - body.extend_from_slice(&leaf); - } - for frame in frames { - body.extend_from_slice(frame.encoded()); - } - let body = Bytes::from(body); - verify_local(&body, &root)?; - Ok((body, digest)) -} - -/// Self-verification covers all bytes of a newly uploaded object. References -/// to old immutable shards remain authenticated by the predecessor selection. -fn verify_local(body: &Bytes, expected: &Root) -> Result<()> { - let root = codec::decode(&body[..HEADER_BYTES])?; - if &root != expected || body.len() as u64 != root.object_bytes { - return Err(Error::Node("bundle index self-verification differs")); - } - let mut offset = HEADER_BYTES as u64; - let mut locators = Vec::new(); - for (id, shard) in root.shards.iter().enumerate() { - let Some(shard) = shard else { continue }; - if shard.extent.object.is_some() { - continue; - } - if shard.extent.offset != offset { - return Err(Error::Node("bundle local shard extents are not canonical")); - } - let bytes = extent_bytes(body, &shard.extent)?; - let leaf = decode_leaf(bytes, shard, id as u8, &root)?; - locators.extend( - leaf.bindings - .into_iter() - .flat_map(|binding| binding.locators) - .filter(|locator| locator.object.is_none()), - ); - offset += shard.extent.bytes; - } - if locators.len() != root.frames.len() { - return Err(Error::Node("bundle local native extent count differs")); - } - for frame in &root.frames { - if frame.offset != offset - || locators.iter().filter(|locator| *locator == frame).count() != 1 - { - return Err(Error::Node("bundle local native extents differ")); - } - extent_bytes(body, frame)?; - offset += frame.bytes; - } - if offset != root.object_bytes { - return Err(Error::Node("bundle has unauthenticated trailing bytes")); - } - Ok(()) + dense::encode(catalog, frames) } fn extent_bytes(body: &Bytes, locator: &Locator) -> Result { @@ -262,13 +120,26 @@ fn extent_bytes(body: &Bytes, locator: &Locator) -> Result { Ok(Bytes::copy_from_slice(bytes)) } +#[cfg(test)] fn decode_leaf(body: Bytes, shard: &Shard, id: u8, root: &Root) -> Result { + decode_leaf_with_histories(body, shard, id, root).map(|(catalog, _)| catalog) +} + +fn decode_leaf_with_histories( + body: Bytes, + shard: &Shard, + id: u8, + root: &Root, +) -> Result<(Catalog, BTreeMap<[u8; 32], history::History>)> { if body.len() as u64 != shard.extent.bytes || *blake3::hash(&body).as_bytes() != *shard.extent.frame_digest.as_bytes() { return Err(Error::Node("bundle catalog shard digest differs")); } - let catalog = super::codec::decode_leaf(&body)?; + let (catalog, histories) = history::decode_leaf(&body)?; + if !root.detached && body.get(..8) == Some(b"\0\0\0\x04CBL3".as_slice()) { + return Err(Error::Node("inline bundle index contains detached history")); + } if catalog.session != root.session || catalog.epoch != root.epoch || catalog.selected_through > root.selected_through @@ -281,11 +152,11 @@ fn decode_leaf(body: Bytes, shard: &Shard, id: u8, root: &Root) -> Result Result { - if body.get(..8) == Some(MAGIC.as_slice()) { + if matches!(body.get(..8), Some(magic) if magic == MAGIC || magic == DENSE_MAGIC) { let header = body .get(..HEADER_BYTES) .ok_or(Error::Node("truncated bundle index header"))?; diff --git a/crates/cellule-runtime/src/node/bundle/index/tests.rs b/crates/cellule-runtime/src/node/bundle/index/tests.rs deleted file mode 100644 index fb058adf..00000000 --- a/crates/cellule-runtime/src/node/bundle/index/tests.rs +++ /dev/null @@ -1,70 +0,0 @@ -use super::*; -use cellule_store::test_support::CountingObjectStore; -use object_store::{memory::InMemory, path::Path}; -use std::sync::Arc; - -#[tokio::test] -async fn selected_shard_budget_is_checked_before_loading_any_leaf() { - let counted = Arc::new(CountingObjectStore::new(Arc::new(InMemory::new()))); - let layout = cellule_ltx::CellStorageLayout::new( - cellule_store::Store::new(counted.clone()), - Path::from("index-admission"), - [9; 16], - ); - let session = SessionId::from_bytes([1; 16]); - let mut shards = vec![None; SHARDS]; - // Both extents are individually bounded. Neither missing leaf should be - // requested when their selected aggregate already exceeds the budget. - for id in [0, 1] { - shards[id] = Some(Shard { - extent: Locator { - object: Some(Digest::from_bytes([id as u8 + 2; 32])), - offset: HEADER_BYTES as u64, - bytes: 3 << 20, - frame_digest: Digest::from_bytes([id as u8 + 4; 32]), - }, - bindings: 1, - }); - } - let root = Root { - session, - epoch: 2, - predecessor: None, - selected_through: 0, - object_bytes: HEADER_BYTES as u64, - shards, - frames: Vec::new(), - }; - let header = codec::encode(&root).unwrap(); - let head = NodeBundleHead { - epoch: 2, - digest: Digest::from_bytes(*blake3::hash(&header).as_bytes()), - selected_through: 0, - }; - layout - .store() - .put_exact( - &layout.node_coverage_bundle_path(session.as_bytes(), 2, head.digest.as_bytes()), - header, - ) - .await - .unwrap(); - counted.reset(); - assert!(matches!( - load(&layout, session, head, Some(&[0, 1].into_iter().collect())).await, - Err(Error::Capacity("bundle selected shard bytes")) - )); - assert_eq!( - counted.requests().len(), - 1, - "only the authenticated header is fetched" - ); - // A point lookup charges only its chosen shard. Other large shards are - // not a reason to reject it; this request reaches the missing origin leaf. - counted.reset(); - assert!(!matches!( - load(&layout, session, head, Some(&[0].into_iter().collect())).await, - Err(Error::Capacity("bundle selected shard bytes")) - )); - assert_eq!(counted.requests().len(), 2); -} diff --git a/crates/cellule-runtime/src/node/bundle/index/tests/admission.rs b/crates/cellule-runtime/src/node/bundle/index/tests/admission.rs new file mode 100644 index 00000000..210c0826 --- /dev/null +++ b/crates/cellule-runtime/src/node/bundle/index/tests/admission.rs @@ -0,0 +1,169 @@ +use super::*; + +#[tokio::test] +async fn aggregate_history_budget_is_checked_before_the_first_history_read() { + let counted = Arc::new(CountingObjectStore::new(Arc::new(InMemory::new()))); + let layout = cellule_ltx::CellStorageLayout::new( + cellule_store::Store::new(counted.clone()), + Path::from("history-admission"), + [9; 16], + ); + let session = SessionId::from_bytes([1; 16]); + let mut bindings = Vec::new(); + let mut histories = BTreeMap::new(); + let mut cells = BTreeSet::new(); + for number in 0_u64..100_000 { + let mut cell = [0; 32]; + cell[..8].copy_from_slice(&number.to_le_bytes()); + if shard(&[9; 16], &cell) != 0 { + continue; + } + let mut binding = history_binding(cell, 1); + let extent = Locator { + object: Some(Digest::from_bytes([8; 32])), + offset: HEADER_BYTES as u64, + bytes: history::MAX_HISTORY_BYTES, + frame_digest: Digest::from_bytes([9; 32]), + }; + binding.locators = vec![extent.clone()]; + histories.insert( + history::pin(&binding).unwrap(), + history::History { + extent, + count: MAX_LOCATORS, + native_bytes: MAX_SUFFIX_BYTES, + loaded: None, + }, + ); + cells.insert(([9; 16], cell)); + bindings.push(binding); + if bindings.len() == 129 { + break; + } + } + assert_eq!(bindings.len(), 129); + bindings + .sort_unstable_by_key(|binding| *binding.control.bundle_binding.unwrap().digest.as_bytes()); + let catalog = Catalog { + session, + epoch: 2, + predecessor: None, + selected_through: 1, + bindings, + index: None, + }; + let body = history::encode_leaf(&catalog, &histories).unwrap(); + let mut shards = vec![None; SHARDS]; + shards[0] = Some(Shard { + extent: Locator { + object: None, + offset: HEADER_BYTES as u64, + bytes: body.len() as u64, + frame_digest: Digest::from_bytes(*blake3::hash(&body).as_bytes()), + }, + bindings: 129, + }); + let root = Root { + detached: true, + session, + epoch: 2, + predecessor: None, + selected_through: 1, + object_bytes: HEADER_BYTES as u64 + body.len() as u64, + shards, + frames: Vec::new(), + }; + let header = codec::encode(&root).unwrap(); + let head = NodeBundleHead { + epoch: 2, + digest: Digest::from_bytes(*blake3::hash(&header).as_bytes()), + selected_through: 1, + }; + let mut object = header.to_vec(); + object.extend_from_slice(&body); + layout + .store() + .put_exact( + &layout.node_coverage_bundle_path(session.as_bytes(), 2, head.digest.as_bytes()), + Bytes::from(object), + ) + .await + .unwrap(); + counted.reset(); + assert!(matches!( + load_cells(&layout, session, head, &cells).await, + Err(Error::Capacity("bundle selected history bytes")) + )); + assert_eq!( + counted.requests().len(), + 2, + "header and selected shard only; missing history objects must not be fetched" + ); +} + +#[tokio::test] +async fn selected_shard_budget_is_checked_before_loading_any_leaf() { + let counted = Arc::new(CountingObjectStore::new(Arc::new(InMemory::new()))); + let layout = cellule_ltx::CellStorageLayout::new( + cellule_store::Store::new(counted.clone()), + Path::from("index-admission"), + [9; 16], + ); + let session = SessionId::from_bytes([1; 16]); + let mut shards = vec![None; SHARDS]; + // Both extents are individually bounded. Neither missing leaf should be + // requested when their selected aggregate already exceeds the budget. + for id in [0, 1] { + shards[id] = Some(Shard { + extent: Locator { + object: Some(Digest::from_bytes([id as u8 + 2; 32])), + offset: HEADER_BYTES as u64, + bytes: 3 << 20, + frame_digest: Digest::from_bytes([id as u8 + 4; 32]), + }, + bindings: 1, + }); + } + let root = Root { + detached: false, + session, + epoch: 2, + predecessor: None, + selected_through: 0, + object_bytes: HEADER_BYTES as u64, + shards, + frames: Vec::new(), + }; + let header = codec::encode(&root).unwrap(); + let head = NodeBundleHead { + epoch: 2, + digest: Digest::from_bytes(*blake3::hash(&header).as_bytes()), + selected_through: 0, + }; + layout + .store() + .put_exact( + &layout.node_coverage_bundle_path(session.as_bytes(), 2, head.digest.as_bytes()), + header, + ) + .await + .unwrap(); + counted.reset(); + assert!(matches!( + load(&layout, session, head, Some(&[0, 1].into_iter().collect())).await, + Err(Error::Capacity("bundle selected shard bytes")) + )); + assert_eq!( + counted.requests().len(), + 1, + "only the authenticated header is fetched" + ); + // A point lookup charges only its chosen shard. Other large shards are + // not a reason to reject it; this request reaches the missing origin leaf. + counted.reset(); + assert!(!matches!( + load(&layout, session, head, Some(&[0].into_iter().collect())).await, + Err(Error::Capacity("bundle selected shard bytes")) + )); + assert_eq!(counted.requests().len(), 2); +} diff --git a/crates/cellule-runtime/src/node/bundle/index/tests/histories.rs b/crates/cellule-runtime/src/node/bundle/index/tests/histories.rs new file mode 100644 index 00000000..260cdb4b --- /dev/null +++ b/crates/cellule-runtime/src/node/bundle/index/tests/histories.rs @@ -0,0 +1,65 @@ +use super::*; + +#[test] +fn detached_history_authenticates_scope_count_bytes_and_the_complete_bounded_array() { + let mut binding = history_binding([4; 32], MAX_LOCATORS); + let session = SessionId::from_bytes([1; 16]); + let body = history::encode(session, 2, &binding).unwrap(); + assert!(body.len() as u64 <= history::MAX_HISTORY_BYTES); + let descriptor = history::History { + extent: Locator { + object: Some(Digest::from_bytes([7; 32])), + offset: HEADER_BYTES as u64, + bytes: body.len() as u64, + frame_digest: Digest::from_bytes(*blake3::hash(&body).as_bytes()), + }, + count: MAX_LOCATORS, + native_bytes: 128 * MAX_LOCATORS as u64, + loaded: None, + }; + assert_eq!( + history::decode(&body, session, 2, &binding, &descriptor).unwrap(), + binding.locators + ); + for (session, epoch) in [(SessionId::from_bytes([2; 16]), 2), (session, 3)] { + assert!(history::decode(&body, session, epoch, &binding, &descriptor).is_err()); + } + let wrong = history_binding([8; 32], MAX_LOCATORS); + assert!(history::decode(&body, session, 2, &wrong, &descriptor).is_err()); + let mut altered = descriptor.clone(); + altered.count -= 1; + assert!(history::decode(&body, session, 2, &binding, &altered).is_err()); + altered = descriptor.clone(); + altered.native_bytes -= 1; + assert!(history::decode(&body, session, 2, &binding, &altered).is_err()); + let mut corrupt = body.to_vec(); + *corrupt.last_mut().unwrap() ^= 1; + assert!(history::decode(&Bytes::from(corrupt), session, 2, &binding, &descriptor).is_err()); + assert!( + history::decode( + &body.slice(..body.len() - 1), + session, + 2, + &binding, + &descriptor + ) + .is_err() + ); + binding.locators.push(binding.locators[0].clone()); + assert!(matches!( + history::encode(session, 2, &binding), + Err(Error::Capacity("bundle history count")) + )); + binding.locators.pop(); + for locator in &mut binding.locators { + locator.bytes = 17 << 10; + } + let oversized = history::encode(session, 2, &binding).unwrap(); + altered = descriptor; + altered.native_bytes = (17 << 10) * MAX_LOCATORS as u64; + altered.extent.frame_digest = Digest::from_bytes(*blake3::hash(&oversized).as_bytes()); + assert!(matches!( + history::decode(&oversized, session, 2, &binding, &altered), + Err(Error::Capacity("bundle history native bytes")) + )); +} diff --git a/crates/cellule-runtime/src/node/bundle/index/tests/legacy.rs b/crates/cellule-runtime/src/node/bundle/index/tests/legacy.rs new file mode 100644 index 00000000..6896d3c9 --- /dev/null +++ b/crates/cellule-runtime/src/node/bundle/index/tests/legacy.rs @@ -0,0 +1,172 @@ +use super::*; + +pub(in crate::node::bundle) fn encode_inline( + catalog: &mut Catalog, + frames: &[cellule_ltx::VerifiedNodeFrame], +) -> Result<(Bytes, Digest)> { + catalog.validate()?; + if frames.len() > MAX_FRAMES { + return Err(Error::Capacity("bundle frame count")); + } + let groups = grouped(catalog); + let mut shards = match &catalog.index { + Some(index) => { + if index.root.session != catalog.session || index.root.epoch != catalog.epoch { + return Err(Error::Fenced); + } + // A partial catalog may modify only a shard it authenticated first. + if groups.keys().any(|id| !index.loaded.contains_key(id)) { + return Err(Error::Node("bundle modifies an unloaded catalog shard")); + } + index.root.shards.clone() + } + None => vec![None; SHARDS], + }; + let changed: BTreeSet = match &catalog.index { + Some(index) => index + .loaded + .iter() + .filter_map(|(id, original)| { + (groups.get(id).map(Vec::as_slice).unwrap_or(&[]) != original.as_slice()) + .then_some(*id) + }) + .collect(), + None => groups.keys().copied().collect(), + }; + // Native extents follow the changed shards; changing numeric offsets never + // changes the leaf encoding length. Compute sizes before filling offsets. + let mut offset = HEADER_BYTES as u64; + for id in &changed { + let rows = groups.get(id).cloned().unwrap_or_default(); + if rows.is_empty() { + shards[usize::from(*id)] = None; + } else { + let bytes = crate::node::bundle::codec::encode_leaf(&leaf(catalog, rows))?.len() as u64; + shards[usize::from(*id)] = Some(Shard { + extent: Locator { + object: None, + offset, + bytes, + frame_digest: Digest::from_bytes([0; 32]), + }, + bindings: groups[id].len(), + }); + offset = offset + .checked_add(bytes) + .ok_or(Error::Capacity("bundle shard bytes"))?; + } + } + let mut native = Vec::with_capacity(frames.len()); + for frame in frames { + let digest = Digest::from_bytes(frame.digest()); + let mut matches = 0; + for binding in &mut catalog.bindings { + let id = binding_shard(binding); + for locator in &mut binding.locators { + if locator.object.is_none() && locator.frame_digest == digest { + if !changed.contains(&id) { + return Err(Error::Node("native frame belongs to an unchanged shard")); + } + locator.offset = offset; + locator.bytes = frame.encoded().len() as u64; + matches += 1; + } + } + } + if matches != 1 { + return Err(Error::Node("bundle frame locator is not unique")); + } + native.push(Locator { + object: None, + offset, + bytes: frame.encoded().len() as u64, + frame_digest: digest, + }); + offset = offset + .checked_add(frame.encoded().len() as u64) + .ok_or(Error::Capacity("bundle bytes"))?; + } + if offset > MAX_BUNDLE_BYTES { + return Err(Error::Capacity("bundle bytes")); + } + let groups = grouped(catalog); + let mut bodies = Vec::new(); + for id in &changed { + if let Some(reference) = &mut shards[usize::from(*id)] { + let body = crate::node::bundle::codec::encode_leaf(&leaf(catalog, groups[id].clone()))?; + if body.len() as u64 != reference.extent.bytes { + return Err(Error::Node("bundle shard size changed during encoding")); + } + reference.extent.frame_digest = Digest::from_bytes(*blake3::hash(&body).as_bytes()); + bodies.push(body); + } + } + let root = Root { + detached: false, + session: catalog.session, + epoch: catalog.epoch, + predecessor: catalog.predecessor, + selected_through: catalog.selected_through, + object_bytes: offset, + shards, + frames: native, + }; + let header = codec::encode(&root)?; + let digest = Digest::from_bytes(*blake3::hash(&header).as_bytes()); + let mut body = Vec::with_capacity(offset as usize); + body.extend_from_slice(&header); + for leaf in bodies { + body.extend_from_slice(&leaf); + } + for frame in frames { + body.extend_from_slice(frame.encoded()); + } + let body = Bytes::from(body); + verify_local(&body, &root)?; + Ok((body, digest)) +} + +/// Self-verification covers all bytes of a newly uploaded object. References +/// to old immutable shards remain authenticated by the predecessor selection. +fn verify_local(body: &Bytes, expected: &Root) -> Result<()> { + let root = codec::decode(&body[..HEADER_BYTES])?; + if &root != expected || body.len() as u64 != root.object_bytes { + return Err(Error::Node("bundle index self-verification differs")); + } + let mut offset = HEADER_BYTES as u64; + let mut locators = Vec::new(); + for (id, shard) in root.shards.iter().enumerate() { + let Some(shard) = shard else { continue }; + if shard.extent.object.is_some() { + continue; + } + if shard.extent.offset != offset { + return Err(Error::Node("bundle local shard extents are not canonical")); + } + let bytes = extent_bytes(body, &shard.extent)?; + let leaf = decode_leaf(bytes, shard, id as u8, &root)?; + locators.extend( + leaf.bindings + .into_iter() + .flat_map(|binding| binding.locators) + .filter(|locator| locator.object.is_none()), + ); + offset += shard.extent.bytes; + } + if locators.len() != root.frames.len() { + return Err(Error::Node("bundle local native extent count differs")); + } + for frame in &root.frames { + if frame.offset != offset + || locators.iter().filter(|locator| *locator == frame).count() != 1 + { + return Err(Error::Node("bundle local native extents differ")); + } + extent_bytes(body, frame)?; + offset += frame.bytes; + } + if offset != root.object_bytes { + return Err(Error::Node("bundle has unauthenticated trailing bytes")); + } + Ok(()) +} diff --git a/crates/cellule-runtime/src/node/bundle/index/tests/mod.rs b/crates/cellule-runtime/src/node/bundle/index/tests/mod.rs new file mode 100644 index 00000000..e7e5e12d --- /dev/null +++ b/crates/cellule-runtime/src/node/bundle/index/tests/mod.rs @@ -0,0 +1,71 @@ +use super::*; +use cellule_store::test_support::CountingObjectStore; +use object_store::{memory::InMemory, path::Path}; +use std::sync::Arc; + +// Codec fixtures authenticate metadata only; they grant no native range proof. +fn history_binding(cell: [u8; 32], count: usize) -> Binding { + let cell_id = crate::identity::CellId::from_bytes(cell); + let incarnation = crate::identity::IncarnationId::from_bytes([14; 16]); + let mut control = Control::initial( + cell_id, + incarnation, + crate::control::Owner { + session: SessionId::from_bytes([1; 16]), + endpoint: "https://codec.internal:8081".into(), + }, + Digest::from_bytes([12; 32]), + 1, + ) + .unwrap(); + control.state = crate::control::ControlState::Serving; + control.root = Some( + crate::control::RootRef::from_ltx( + cell_id, + incarnation, + cellule_ltx::RootRef { + cell, + incarnation: [14; 16], + digest: [2; 32], + position: cellule_ltx::Position { + txid: 1, + checksum: cellule_ltx::types::CHECKSUM_FLAG | 3, + }, + commit_sequence: 1, + }, + ) + .unwrap(), + ); + control.bundle_binding = Some(BundleBindingRef { + session: SessionId::from_bytes([1; 16]), + epoch: 2, + digest: Digest::from_bytes(*blake3::hash(&cell).as_bytes()), + }); + Binding { + application: crate::identity::ApplicationId::from_bytes([9; 16]), + first_commit: 2, + control, + phase: BindingPhase::Open, + terminal: None, + selected_sequence: 1, + selected_commit: 2, + selected_position: cellule_ltx::Position { + txid: 2, + checksum: cellule_ltx::types::CHECKSUM_FLAG | 4, + }, + locators: vec![ + Locator { + object: Some(Digest::from_bytes([5; 32])), + offset: HEADER_BYTES as u64, + bytes: 128, + frame_digest: Digest::from_bytes([6; 32]) + }; + count + ], + } +} + +mod admission; +mod histories; +mod legacy; +pub(in crate::node::bundle) use legacy::encode_inline; diff --git a/crates/cellule-runtime/src/node/bundle/mod.rs b/crates/cellule-runtime/src/node/bundle/mod.rs index b3fa0e1a..1a81bd5e 100644 --- a/crates/cellule-runtime/src/node/bundle/mod.rs +++ b/crates/cellule-runtime/src/node/bundle/mod.rs @@ -52,7 +52,8 @@ mod tests; pub(crate) const MAX_BUNDLE_BYTES: u64 = 4 << 20; const MAX_BINDINGS: usize = 4_096; -const MAX_LOCATORS: usize = 32; +const MAX_LOCATORS: usize = 256; +const MAX_INLINE_LOCATORS: usize = 32; const MAX_FRAMES: usize = 64; // Verification retains at most one Cell suffix, independently of the number // of historical objects referenced by its locators. @@ -185,6 +186,7 @@ impl Catalog { let mut previous = None; let mut scopes = std::collections::HashSet::new(); for binding in &self.bindings { + index::validate_deferred(self, binding)?; binding.control.encode()?; let pin = binding .control diff --git a/crates/cellule-runtime/src/node/bundle/proof.rs b/crates/cellule-runtime/src/node/bundle/proof.rs index 8bbfd449..3d8b9f3b 100644 --- a/crates/cellule-runtime/src/node/bundle/proof.rs +++ b/crates/cellule-runtime/src/node/bundle/proof.rs @@ -41,13 +41,10 @@ impl NodeDirectory { return Err(Error::Fenced); } let head = head.ok_or(Error::PendingPublication)?; - let shards = [index::shard( - authority.layout().application_id(), - value.cell.as_bytes(), - )] - .into_iter() - .collect(); - let mut catalog = store::load_catalog_shards(&self.layout, session, head, &shards).await?; + let cells = [(*authority.layout().application_id(), *value.cell.as_bytes())] + .into_iter() + .collect(); + let mut catalog = store::load_catalog_cells(&self.layout, session, head, &cells).await?; let binding = catalog.binding_mut(pin.digest)?.clone(); if binding.phase == BindingPhase::Provisional { return Err(Error::PendingPublication); diff --git a/crates/cellule-runtime/src/node/bundle/selection.rs b/crates/cellule-runtime/src/node/bundle/selection.rs index 1a4a29f3..3272c520 100644 --- a/crates/cellule-runtime/src/node/bundle/selection.rs +++ b/crates/cellule-runtime/src/node/bundle/selection.rs @@ -21,15 +21,15 @@ impl NodeDirectory { if frames.is_empty() || frames.len() > MAX_FRAMES { return Err(Error::Capacity("bundle frame count")); } - let shards = frames + let cells = frames .iter() .map(|frame| { let scope = frame.scope(); - index::shard(&scope.application, &scope.cell) + (scope.application, scope.cell) }) .collect(); let mut catalog = - store::load_catalog_shards(&self.layout, observed.advertisement.session, head, &shards) + store::load_catalog_cells(&self.layout, observed.advertisement.session, head, &cells) .await?; let mut consumed = 0_usize; for assignment in assignments { @@ -121,31 +121,36 @@ impl NodeDirectory { now_ms: i64, ) -> Result<(VersionedNodeAdvertisement, Vec)> { lease.check()?; - let shards = prepared + let cells = prepared .catalog .bindings .iter() + .filter(|binding| { + binding + .locators + .iter() + .any(|locator| locator.object.is_none()) + }) .map(|binding| { - index::shard( - binding.application.as_bytes(), - binding.control.cell.as_bytes(), + ( + *binding.application.as_bytes(), + *binding.control.cell.as_bytes(), ) }) .collect(); - let catalog = store::load_catalog_shards( + let catalog = store::load_catalog_cells( &self.layout, prepared.catalog.session, prepared.head, - &shards, + &cells, ) .await?; let mut proofs = Vec::new(); for binding in catalog.bindings { - if !binding - .locators - .iter() - .any(|locator| locator.object == Some(prepared.head.digest)) - { + if !cells.contains(&( + *binding.application.as_bytes(), + *binding.control.cell.as_bytes(), + )) { continue; } verify_binding( diff --git a/crates/cellule-runtime/src/node/bundle/store.rs b/crates/cellule-runtime/src/node/bundle/store.rs index b0facd29..facd613f 100644 --- a/crates/cellule-runtime/src/node/bundle/store.rs +++ b/crates/cellule-runtime/src/node/bundle/store.rs @@ -115,13 +115,13 @@ pub(super) async fn load_catalog( index::load(layout, session, head, None).await } -pub(super) async fn load_catalog_shards( +pub(super) async fn load_catalog_cells( layout: &cellule_ltx::CellStorageLayout, session: SessionId, head: NodeBundleHead, - shards: &std::collections::BTreeSet, + cells: &std::collections::BTreeSet, ) -> Result { - index::load(layout, session, head, Some(shards)).await + index::load_cells(layout, session, head, cells).await } pub(super) async fn load_legacy_catalog( @@ -186,13 +186,10 @@ pub(crate) async fn ensure_enrollment( return Err(Error::Fenced); } let head = node.bundle.ok_or(Error::PendingPublication)?; - let shards = [index::shard( - layout.application_id(), - control.cell.as_bytes(), - )] - .into_iter() - .collect(); - let mut catalog = load_catalog_shards(layout, pin.session, head, &shards).await?; + let cells = [(*layout.application_id(), *control.cell.as_bytes())] + .into_iter() + .collect(); + let mut catalog = load_catalog_cells(layout, pin.session, head, &cells).await?; let binding = catalog.binding_mut(pin.digest)?; if head.epoch != pin.epoch || binding.phase != BindingPhase::Provisional @@ -236,13 +233,10 @@ pub(crate) async fn ensure_departure( return Err(Error::Fenced); } let head = head.ok_or(Error::PendingPublication)?; - let shards = [index::shard( - layout.application_id(), - control.cell.as_bytes(), - )] - .into_iter() - .collect(); - let mut catalog = load_catalog_shards(layout, pin.session, head, &shards).await?; + let cells = [(*layout.application_id(), *control.cell.as_bytes())] + .into_iter() + .collect(); + let mut catalog = load_catalog_cells(layout, pin.session, head, &cells).await?; if pin.epoch != catalog.epoch { return Err(Error::Fenced); } diff --git a/crates/cellule-runtime/src/node/bundle/tests/index/checkpoint.rs b/crates/cellule-runtime/src/node/bundle/tests/index/checkpoint.rs index df74c44e..6653885e 100644 --- a/crates/cellule-runtime/src/node/bundle/tests/index/checkpoint.rs +++ b/crates/cellule-runtime/src/node/bundle/tests/index/checkpoint.rs @@ -46,8 +46,8 @@ async fn checkpoint_uses_the_exact_materialized_proof_without_scanning_siblings_ .collect(); assert_eq!( reads.len(), - 2, - "checkpoint should read only its authenticated header and chosen shard" + 3, + "checkpoint reads its authenticated header, chosen shard and exact history; no native frames" ); assert_eq!(f.count.put_requests(), 2); f.node = checkpoint; diff --git a/crates/cellule-runtime/src/node/bundle/tests/index/compatibility.rs b/crates/cellule-runtime/src/node/bundle/tests/index/compatibility.rs index 21e6e14e..cfbd4b53 100644 --- a/crates/cellule-runtime/src/node/bundle/tests/index/compatibility.rs +++ b/crates/cellule-runtime/src/node/bundle/tests/index/compatibility.rs @@ -1,5 +1,79 @@ use super::*; +#[tokio::test] +async fn inline_indexed_suffix_migrates_to_detached_history_with_the_same_pin() { + let mut f = Fixture::new().await; + let mut cell = f.cell(4).await; + let (_, frames, assignment) = f.append(&mut cell, 2); + let proposal = f + .directory + .prepare_node_bundle(&f.node, &frames, &[assignment], NOW) + .await + .unwrap(); + f.node = f + .directory + .select_node_bundle(&f.node, &proposal, &f.lease, Limits::default(), NOW) + .await + .unwrap() + .0; + let original = f.node.advertisement().bundle_head().unwrap(); + let mut catalog = load_catalog(&f.layout, head_session(&f), original) + .await + .unwrap(); + catalog.index = None; + catalog.predecessor = Some(original.digest()); + let (body, digest) = catalog_index::encode_inline(&mut catalog, &[]).unwrap(); + assert_eq!(&body[..8], b"\0\0\0\x04CNB2"); + let legacy = PreparedNodeBundle { + original: Some(original), + head: NodeBundleHead { + epoch: EPOCH, + digest, + selected_through: original.selected_through(), + }, + body, + catalog, + }; + f.layout + .store() + .put_exact( + &f.layout + .node_coverage_bundle_path(&[1; 16], EPOCH, digest.as_bytes()), + legacy.body.clone(), + ) + .await + .unwrap(); + f.node = f + .directory + .select_catalog(&f.node, &legacy, NOW) + .await + .unwrap(); + let old_pin = cell.control.value().bundle_binding.unwrap(); + assert_eq!( + f.directory + .load_bundle_coverage(&cell.authority, &cell.control, Limits::default()) + .await + .unwrap() + .commit_sequence(), + 2 + ); + let (_, frames, assignment) = f.append(&mut cell, 3); + let proposal = f + .directory + .prepare_node_bundle(&f.node, &frames, &[assignment], NOW) + .await + .unwrap(); + assert_eq!(&proposal.body[..8], b"\0\0\0\x04CNB3"); + let (_, proofs) = f + .directory + .select_node_bundle(&f.node, &proposal, &f.lease, Limits::default(), NOW) + .await + .unwrap(); + assert_eq!(proofs[0].binding(), old_pin); + assert_eq!(proofs[0].locator_count(), 2); + assert_eq!(proofs[0].commit_sequence(), 3); +} + #[tokio::test] async fn legacy_selected_catalog_is_read_and_migrated_without_changing_the_cell_pin() { let mut f = Fixture::new().await; @@ -49,7 +123,7 @@ async fn legacy_selected_catalog_is_read_and_migrated_without_changing_the_cell_ .prepare_node_bundle(&f.node, &frames, &[assigned], NOW) .await .unwrap(); - assert_eq!(&proposal.body[..8], b"\0\0\0\x04CNB2"); + assert_eq!(&proposal.body[..8], b"\0\0\0\x04CNB3"); let (_, proofs) = f .directory .select_node_bundle(&f.node, &proposal, &f.lease, Limits::default(), NOW) @@ -118,15 +192,15 @@ async fn indexed_shard_corruption_and_missing_reused_shards_fail_closed() { .collect(); assert_eq!( reads.len(), - 3, - "a reused shard does not require the old root header" + 4, + "a reused shard/history does not require the old root header" ); assert_eq!( reads .iter() .filter(|read| read.location == old_path.as_ref()) .count(), - 2 + 3 ); // The head/header stays unchanged; the independently authenticated reused // shard must still detect a corrupt provider payload in the old object. diff --git a/crates/cellule-runtime/src/node/bundle/tests/index/copy_on_write.rs b/crates/cellule-runtime/src/node/bundle/tests/index/copy_on_write.rs index 0443ed43..909af294 100644 --- a/crates/cellule-runtime/src/node/bundle/tests/index/copy_on_write.rs +++ b/crates/cellule-runtime/src/node/bundle/tests/index/copy_on_write.rs @@ -51,7 +51,7 @@ async fn updating_one_of_a_thousand_cells_does_not_rewrite_the_complete_inventor ); assert_eq!( catalog_reads.len(), - 3, - "one header, one shard and one exact native frame" + 4, + "one header, one shard, one history and one exact native frame" ); } diff --git a/crates/cellule-runtime/src/node/bundle/tests/index/density.rs b/crates/cellule-runtime/src/node/bundle/tests/index/density.rs new file mode 100644 index 00000000..388941f2 --- /dev/null +++ b/crates/cellule-runtime/src/node/bundle/tests/index/density.rs @@ -0,0 +1,210 @@ +use super::*; + +#[tokio::test] +async fn dense_history_retains_215_exact_commands_before_checkpoint_and_cold_restores() { + let mut f = Fixture::new().await; + let mut cell = f.cell(4).await; + inventory(&mut f, &cell, 2_000).await; + let mut latest = None; + let mut segments = Vec::new(); + let mut metadata_bytes = 0; + for commit in 2..=216 { + let (cuts, frames, assignment) = f.append(&mut cell, commit); + segments.extend(cuts.segments); + let proposal = f + .directory + .prepare_node_bundle(&f.node, &frames, &[assignment], NOW) + .await + .unwrap(); + metadata_bytes = proposal.body.len() + - frames + .iter() + .map(|frame| frame.encoded().len()) + .sum::(); + let (selected, mut proofs) = f + .directory + .select_node_bundle(&f.node, &proposal, &f.lease, Limits::default(), NOW) + .await + .unwrap(); + f.node = selected; + latest = proofs.pop(); + } + let proof = latest.unwrap(); + assert_eq!(proof.commit_sequence(), 216); + assert_eq!(proof.locator_count(), 215); + // A following local command is deliberately outside the selected endpoint. + let _unselected = f.append(&mut cell, 217); + f.count.reset(); + let root = f.publisher(&cell).materialize_bundle(&proof).await.unwrap(); + let puts = f.count.put_requests(); + eprintln!( + "dense checkpoint: catalog_bindings=2000 commands=215 materialization_puts={puts} selection_metadata_bytes={metadata_bytes}" + ); + assert_eq!(puts, 4, "the small-tail cost must be measured, not assumed"); + let direct = cell + .replica + .prepare( + Some(&proof.base().unwrap()), + &cellule_ltx::CaptureBatch { + segments, + position: proof.position(), + timing: Default::default(), + }, + 216, + 1, + ) + .await + .unwrap(); + assert_eq!( + root, + direct.root(), + "same canonical root digest and byte image as original native captures" + ); + let restored = f.scratch.path().join("dense-cold.sqlite"); + cell.replica + .open_root(&root) + .await + .unwrap() + .restore(&restored) + .await + .unwrap(); + let db = rusqlite::Connection::open(&restored).unwrap(); + let count: u64 = db + .query_row("SELECT count(*) FROM outcomes", [], |row| row.get(0)) + .unwrap(); + assert_eq!( + count, 216, + "seed plus every selected command, no later command" + ); + let outcomes: Vec<(String, String)> = db + .prepare("SELECT request, result FROM outcomes ORDER BY request") + .unwrap() + .query_map([], |row| Ok((row.get(0)?, row.get(1)?))) + .unwrap() + .collect::>() + .unwrap(); + let mut expected: Vec<_> = (2..=216) + .map(|commit| (format!("request-{commit}"), format!("result-{commit}"))) + .collect(); + expected.push(("seed".into(), "original".into())); + expected.sort(); + assert_eq!(outcomes, expected); + f.directory + .checkpoint_bundle_cell(&f.node, &cell.authority, &proof, Limits::default(), NOW) + .await + .unwrap(); +} + +#[tokio::test] +async fn a_point_update_neither_fetches_nor_rewrites_a_siblings_detached_history() { + let mut f = Fixture::new().await; + let mut a = f.cell(4).await; + let wanted = catalog_index::shard(&[9; 16], &[4; 32]); + let application = (0_u64..10_000) + .map(|number| { + let mut application = [0; 16]; + application[..8].copy_from_slice(&number.to_le_bytes()); + application + }) + .find(|application| catalog_index::shard(application, &[4; 32]) == wanted) + .unwrap(); + let mut b = f.cell_for_application(4, application).await; + for cell in [&mut a, &mut b] { + let (_, frames, assignment) = f.append(cell, 2); + let proposal = f + .directory + .prepare_node_bundle(&f.node, &frames, &[assignment], NOW) + .await + .unwrap(); + f.node = f + .directory + .select_node_bundle(&f.node, &proposal, &f.lease, Limits::default(), NOW) + .await + .unwrap() + .0; + } + let head = f.node.advertisement().bundle_head().unwrap(); + let cells = [([9; 16], [4; 32])].into_iter().collect(); + let catalog = store::load_catalog_cells(&f.layout, head_session(&f), head, &cells) + .await + .unwrap(); + let history = + catalog_index::history_extent(&catalog, b.control.value().bundle_binding.unwrap().digest) + .unwrap(); + let object = history.object.unwrap(); + let path = f + .layout + .node_coverage_bundle_path(&[1; 16], EPOCH, object.as_bytes()); + let (body, _) = f + .layout + .store() + .get_with_etag_bounded(&path, MAX_BUNDLE_BYTES) + .await + .unwrap(); + let mut corrupt = body.to_vec(); + corrupt[history.offset as usize + 8] ^= 1; + f.layout + .store() + .put_overwrite(&path, Bytes::from(corrupt)) + .await + .unwrap(); + assert!( + f.directory + .load_bundle_coverage(&b.authority, &b.control, Limits::default()) + .await + .is_err() + ); + f.count.reset(); + let (_, frames, assignment) = f.append(&mut a, 3); + let proposal = f + .directory + .prepare_node_bundle(&f.node, &frames, &[assignment], NOW) + .await + .unwrap(); + let (selected, proofs) = f + .directory + .select_node_bundle(&f.node, &proposal, &f.lease, Limits::default(), NOW) + .await + .unwrap(); + assert_eq!(proofs.len(), 1); + assert_eq!( + proofs[0].binding(), + a.control.value().bundle_binding.unwrap() + ); + assert_eq!(object, head.digest()); + assert_eq!( + f.count + .requests() + .iter() + .filter(|read| read.location == path.as_ref()) + .count(), + 2, + "only the previous header and shared shard are read; no sibling history" + ); + let selected_catalog = store::load_catalog_cells( + &f.layout, + head_session(&f), + selected.advertisement().bundle_head().unwrap(), + &cells, + ) + .await + .unwrap(); + assert_eq!( + catalog_index::history_extent( + &selected_catalog, + b.control.value().bundle_binding.unwrap().digest + ), + Some(history) + ); + // Complete reconstruction inventory still refuses the corrupt sibling; + // point selection grants no cross-Cell collection or drain authority. + assert!( + load_catalog( + &f.layout, + head_session(&f), + selected.advertisement().bundle_head().unwrap() + ) + .await + .is_err() + ); +} diff --git a/crates/cellule-runtime/src/node/bundle/tests/index/mod.rs b/crates/cellule-runtime/src/node/bundle/tests/index/mod.rs index 33e23864..79f7654d 100644 --- a/crates/cellule-runtime/src/node/bundle/tests/index/mod.rs +++ b/crates/cellule-runtime/src/node/bundle/tests/index/mod.rs @@ -43,4 +43,5 @@ mod checkpoint; mod cohort; mod compatibility; mod copy_on_write; +mod density; mod inventory; diff --git a/crates/cellule-runtime/src/node/bundle/tests/ranges.rs b/crates/cellule-runtime/src/node/bundle/tests/ranges.rs index 1759e083..8e7a186a 100644 --- a/crates/cellule-runtime/src/node/bundle/tests/ranges.rs +++ b/crates/cellule-runtime/src/node/bundle/tests/ranges.rs @@ -216,8 +216,17 @@ async fn corrupt_origin_dependency_and_overlapping_ranges_fail_closed() { .layout .node_coverage_bundle_path(&[1; 16], EPOCH, prepared.head.digest.as_bytes()); let mut corrupt = prepared.body.to_vec(); - let last = corrupt.last_mut().unwrap(); - *last ^= 1; + // Native bodies precede the detached history in CNB3. Corrupt the actual + // retained native extent, rather than an older superseded history page. + let native_offset = prepared + .catalog + .bindings + .iter() + .flat_map(|binding| &binding.locators) + .find(|locator| locator.object.is_none()) + .unwrap() + .offset as usize; + corrupt[native_offset + 8] ^= 1; f.layout .store() .inner() diff --git a/docs/bundle-coverage-implementation.md b/docs/bundle-coverage-implementation.md index ffc0f53b..638c0a81 100644 --- a/docs/bundle-coverage-implementation.md +++ b/docs/bundle-coverage-implementation.md @@ -76,56 +76,103 @@ unknown fields are rejected. Old readers must not operate on those records. Use a fresh development prefix and initialize the lane before native issuance; there is no live-format migration or same-boot bundle-epoch rotation. -New `CNB2` objects contain a fixed authenticated root index, changed catalog -shards and native extents under the existing -`cells/v1/node-logs///coverage/v1/.cnb` path. The selected -head digest authenticates the 32 KiB canonical header; the header authenticates -all shard ranges and new native extents. Unchanged shard references retain their -original object, offset, length and digest. A Cell lookup needs one header and -one shard, then its exact native ranges; it never follows a predecessor chain. -Both the application and Cell ID determine the shard. Complete maintenance -inventory still reads and verifies every shard, retaining one bounded shard at -a time and a duplicate-pin set capped at 4,096 entries. -Hot cohort lookups preflight the sum of their chosen shard lengths against a -4 MiB encoded-metadata budget before leaf I/O; a point lookup charges only its -own shard. This bounds protocol input, not total heap usage or host admission. - -The `CNB1` complete-catalog reader remains available. Its selected digest still -authenticates the whole object. A new update migrates that catalog to `CNB2` -without changing Cell pins. Older binaries cannot read `CNB2`; upgrade all -recovery consumers before selecting this format. This does not migrate unbound -live actor activations or enable bundle-based responses. +New `CNB3` objects contain a fixed authenticated root index, changed catalog +shards, native extents and detached histories under the existing +`cells/v1/node-logs///coverage/v1/.cnb` path. The head digest +authenticates the 32 KiB header; shard digests authenticate each binding row and +its exact history extent; history digests authenticate every native reference. +New histories share the one immutable upload. Old native bodies are never copied +into a new history. Unchanged shards and histories keep exact object/range/digest +references without walking a predecessor chain. + +A point lookup fetches one header, one shard and only the requested Cell's +history, then its exact native frames. A shared shard keeps unrelated histories +as authenticated references, including when those bodies are unavailable. +Selection verifies every participating Cell's origin and native suffix before +CAS; point selection grants no sibling drain or collection authority. +Complete maintenance inventory still verifies every shard, retaining one +bounded shard at a time and at most 4,096 duplicate-pin entries. An unresolved +history remains a drain obligation. + +Hot lookups preflight selected shard lengths against 4 MiB before leaf reads, +then preflight the selected histories plus those shards against that same bound +before the first history read. This bounds protocol input, not total heap usage +or host admission. Hydration alone does not rewrite a shard; modifications to an +unloaded history reject. + +`CNB1` complete catalogs and `CNB2` inline indexed shards remain readable, with +their original 32-inline-locator bound. Updating them selects `CNB3` without +changing Cell pins; unchanged inline shards can remain referenced. Older binaries +cannot read `CNB3`. Upgrade every recovery consumer before selecting it, using a +fresh development prefix. This does not migrate unbound live actor activations +or enable bundle responses. | Development bound | Value | | --- | ---: | | Immutable object or individual catalog shard bytes | 4 MiB | -| Encoded catalog shards loaded for one cohort | 4 MiB, checked before leaf reads | +| Selected shard plus history encoded metadata | 4 MiB, checked before the respective reads | | Authenticated index header | 32 KiB / 256 shard references | | Original bindings per boot/epoch | 4,096 | | Native frames per selection | 64 | -| Uncheckpointed locators per Cell | 32 | -| Encoded suffix bytes verified per Cell | 4 MiB | +| Detached exact frame references per Cell | 256 | +| One encoded history | 32 KiB | +| Legacy inline locators per Cell | 32 | +| Native suffix bytes verified per Cell | 4 MiB | | Distinct base-root origin dependencies verified per Cell | 65,536 | -Exhaustion rejects further bundle preparation while retaining the last proof. -These are safety ceilings, **not performance-qualified policies**. New selections rewrite only changed shards. Old locator bodies are still -reverified. The indexed point lookup and copy-on-write catalog are implemented; -the 32-locator bound, materializer policy and full lifecycle cost do not yet meet -the current conditional 215-command checkpoint cost constraint in the [authority decision](bundle-coverage-proof.md#quantified-checkpoint-constraint). -The caller owns host memory/native-job admission; automatic materializer budget, -fairness and cancellation/drain ownership are not integrated. +Exhaustion rejects preparation while retaining the last proof. These are safety +ceilings, **not performance-qualified policies**. Old native bodies are still +reverified. The 215-command checkpoint density is now represented and measured +for a small-image fixture; total lifecycle cost is not qualified. Host memory, +native-job admission, materializer fairness and joined scheduling remain caller +obligations. See the [node performance design](../crates/cellule-runtime/docs/write-performance-design.md). ## Verification and remaining delivery +The detached-history regression first fails at the original 32-reference +ceiling. With the new representation, one actual Cell among a **2,000-binding +catalog** selects 215 commands without checkpoint and cold-restores its seed and +all 215 outcomes while excluding a later unselected command. The other bindings +are metadata fixtures; this is not 2,000 active writers or node capacity evidence. +Density alone initially costs **220 materialization PUTs**. Streaming verified +rows through the existing admitted coalescer reduces it to **four**, retaining +the 256 KiB changed-page bound. A public file-backed LTX test exercises more than +256 KiB of aggregate native input and uses two LTX PUTs, plus the runtime's lineage +and Cell CAS. The 256-reference boundary also costs four in this fixture; an +oversized changed image preserves the ordinary bundle fallback. + +The new one-Cell update among 1,000 bindings uses **35,416 metadata bytes**, versus +35,223 with inline `CNB2`. Recovery uses four coverage-object ranges: header, +shard, history and native frame. Prefix checkpoint uses three ranges with no +native-body reads. The extra history read buys selective dense lookup; it is not +a claimed read TPS improvement. One shared shard test corrupts a sibling history: +the requested Cell still selects without fetching or rewriting it, while a +complete reconstruction load rejects the missing/corrupt sibling. Additional +codec tests cover original pin/session/epoch, exact count/native-byte claims, +truncation, the 256-reference limit and aggregate admission before history I/O. +Both legacy complete catalogs and inline indexed suffixes migrate with the same +Cell pin. All raw logs remain outside Git. + +The frozen detached-history and streaming-materialization source passed all +twelve contributor checks: **1,908 workspace tests passed, 38 ignored; 60 local +LTX tests passed**. Its 39 focused bundle/index tests include the 215-command +exact-root and outcome regression. The file-backed recovery test also passes. +These results verify correctness and component work; the latest source has no +new end-to-end TPS measurement. Ordinary application responses still use +per-Cell publication, and production bundle ACK integration remains unfinished. + +The following counts describe the earlier inline-index snapshot, not the latest +application TPS: + The index regression reproduces a **783,146-byte** metadata rewrite for one update among 1,000 Cells on the old source. The same test measures **35,223 bytes** -with `CNB2` (95.5% less), retains one immutable PUT plus one node CAS, and recovers +with the original `CNB2` (95.5% less), retains one immutable PUT plus one node CAS, and recovers through three coverage-object ranges: header, chosen shard and native frame. An exact materialized-prefix checkpoint among those 1,000 Cells falls from **253 coverage-object reads to two**, without reading old native frames. A 64-Cell checkpoint cohort selects all completed roots with **two PUTs**, keeping a hot Cell's newer suffix; an unmaterialized participant rejects the whole cohort -without a partial CAS. Small independent materialization now uses the canonical +without a partial CAS. That small independent materialization used the canonical native coalescer and pack: **256 PUTs instead of 320** for those 64 roots, or four per root. A 32-locator suffix falls from **37 PUTs to four**, and cold restore includes every selected outcome while excluding the later unselected command. @@ -134,7 +181,7 @@ The aggregate native rows, indexes and pack headers must fit the existing The [updated cost calculation](bundle-coverage-proof.md#quantified-checkpoint-constraint) requires at least 215 commands per Cell checkpoint if that four-PUT path and 64/64 cohorts hold, before retries or maintenance. Real larger checkpoints must -be measured; the 32-locator safety ceiling does not satisfy this constraint. +be measured; that original 32-locator safety ceiling did not satisfy this constraint. Additional tests cover legacy migration with the same Cell pin, a reused shard without reading its old header, corrupt and missing reused shards, and refusal diff --git a/docs/bundle-coverage-proof.md b/docs/bundle-coverage-proof.md index 1feda875..4511e9fd 100644 --- a/docs/bundle-coverage-proof.md +++ b/docs/bundle-coverage-proof.md @@ -189,7 +189,7 @@ five per root, while the new shared checkpoint adds two PUTs for the entire per Cell checkpoint**, before compaction, retries or collection. With an actual three-PUT materializer that bound would be 162; with four it would be 215 once the shared checkpoint is included. These are conditional calculations, not a -qualified application result. At one command per capture, the current +qualified application result. At one command per capture, the original 32-locator limit cannot meet even the optimistic bound. Raising that limit without admitted file-backed metadata and bounded cold/read cost is insufficient. @@ -201,8 +201,31 @@ If that four-PUT materialization holds at the eventual checkpoint spacing, `2 / 64 + (4 + 2 / 64) / commands_per_checkpoint <= 0.05` requires **215 commands per Cell checkpoint**, before compaction, retries or collection. This remains conditional: checkpoint density, larger-tail cost and all-ACK cold/read bounds -must be measured after the file-backed locator work. The present 32-locator -ceiling still cannot satisfy it for one-command captures. +must be measured after the file-backed locator work. That original 32-locator +ceiling could not satisfy it for one-command captures. + +The next development representation detaches authenticated histories from +binding shards and retains at most 256 exact frame references per Cell, with +a 32 KiB encoded-history bound and the unchanged 4 MiB native-suffix bound. +Only the requested histories are fetched, so increasing density does not load +all of a hot Cell's sibling histories. A regression over a 2,000-binding catalog +retains 215 actual commands without checkpoint and cold-restores every selected +outcome, excluding the later local command. + +Density alone initially materialized that small 215-command tail with **220 +PUTs**, disproving the four-PUT assumption at that spacing. Streaming the fully +validated original rows through the existing bounded coalescer reduces it to +**four PUTs** without raising the changed-image bound. A file-backed public LTX +test exercises aggregate inputs exceeding 256 KiB and uses two LTX PUTs; runtime +lineage and Cell CAS account for the other two. The 256-reference boundary also +uses four PUTs in this small-image fixture. Large changed images retain their +measured ordinary fallback and may cost more. + +The 215-command conditional cost model is now representable and measured for +this fixture. It is still not qualified at 64-command/64-root production cohorts, +and excludes compaction, retries and collection. The new node target and remaining +admission, ACK, recovery and read gates are in the +[runtime design](../crates/cellule-runtime/docs/write-performance-design.md). At the uniform 1,000-Cell bucket target of 2,000 commands/s, 160 commands per Cell span approximately 80 seconds; at 15,000 Fleet commands/s they span about diff --git a/docs/write-performance-delivery.md b/docs/write-performance-delivery.md index fc9c44b2..cff87a12 100644 --- a/docs/write-performance-delivery.md +++ b/docs/write-performance-delivery.md @@ -23,6 +23,14 @@ ordinary actor responses still use the previous publication path. Canonical small-tail materialization also reduces the 64-root cohort from 320 to 256 PUTs and a 32-locator suffix from 37 PUTs to four, with exact cold restore. +Detached histories now retain 256 exact references per Cell under the original +4 MiB native-suffix bound, fetching only requested histories. A small-image +215-command regression among 2,000 catalog bindings initially materializes with +220 PUTs; the streaming bounded coalescer reduces it to four. These are component +measurements, not 2,000-writer or application TPS qualification. The +[runtime design](../crates/cellule-runtime/docs/write-performance-design.md) records +the 8-vCPU/16-GiB node target: 2,000 Cells, 10K write TPS and 50K read TPS. + ## Delivered behavior | Change | Measurable result | Preserved contract | @@ -57,7 +65,7 @@ collection paths. There is no legacy decoding or automatic migration. | M1 | Packs, inline leaves and bounded compaction spooling delivered | Ordinary append meets four PUTs; composed compaction needs five. Paired cold/sparse-read guardrail unverified | | M2 | Bounded file-backed shared publication coordinator implemented; exact scope, restore, cancellation, minimum-budget and dormant-sibling retention checks added | Latest three active-Fleet windows cost 5.229–5.433 PUTs/command; two fail the debt trend. Three per-Cell authority PUTs remain; M4 is required | | M3 | Signed 512-sequence/five-second grants, bounded local registry, lifecycle gate and signed HTTP fixture implemented | Full isolated checks and native lifecycle suite pass; latest three active-Fleet windows cost 0.0138 enrollment GETs/command. The 15K target diagnostic fails delivery and warm audit | -| M4 | [Connected protocol APIs](bundle-coverage-implementation.md), authenticated copy-on-write catalog shards, exact prefix/cohort checkpoints and streamed complete inventory | Actor response/read integration, admitted materializer scheduling, checkpoint density, failed-node issued-suffix recovery and bundle collection remain incomplete; bundle ACKs disabled | +| M4 | [Connected protocol APIs](bundle-coverage-implementation.md), authenticated copy-on-write catalog shards, exact prefix/cohort checkpoints and streamed complete inventory | Actor response/read integration, admitted materializer scheduling, production checkpoint policy, failed-node issued-suffix recovery and bundle collection remain incomplete; bundle ACKs disabled | | M5 | Three paired low-rate Fleet repetitions and target diagnostics with exact ACK audits delivered | Publication stability and target delivery fail; qualified capacity, read/failure/overload matrix and absolute/relative parity remain unverified | ## Shared publication checkpoint From 720bd4e1b2246b0cc8667d6aa03b1a51c5c76aed Mon Sep 17 00:00:00 2001 From: forhappy Date: Wed, 7 Oct 2026 18:25:38 -0700 Subject: [PATCH 021/102] Recover selected bundle prefixes with complete follower tails --- crates/cellule-ltx/docs/recovery.md | 6 + crates/cellule-ltx/src/replica/mod.rs | 21 + .../docs/failover-and-followers.md | 18 + .../docs/write-performance-design.md | 8 + .../src/control/authority/mod.rs | 8 + crates/cellule-runtime/src/control/mod.rs | 5 +- crates/cellule-runtime/src/control/tests.rs | 34 ++ .../src/node/advertisement/mod.rs | 8 + .../src/node/bundle/index/io.rs | 35 ++ .../src/node/bundle/index/mod.rs | 2 +- crates/cellule-runtime/src/node/bundle/mod.rs | 1 + .../cellule-runtime/src/node/bundle/proof.rs | 75 ++-- .../src/node/bundle/recovery.rs | 142 +++++++ .../src/node/bundle/tests/mod.rs | 1 + .../src/node/bundle/tests/recovery.rs | 396 ++++++++++++++++++ .../cellule-runtime/src/node/directory/mod.rs | 1 + .../src/node/directory/recovery.rs | 9 + .../node/log/{recovery.rs => recovery/mod.rs} | 183 +------- .../src/node/log/recovery/streaming.rs | 301 +++++++++++++ .../src/node/log_recovery/mod.rs | 132 +++++- .../src/recovery/manifest/mod.rs | 25 +- .../src/recovery/manifest/tests.rs | 135 ++++++ docs/bundle-coverage-implementation.md | 29 +- 23 files changed, 1336 insertions(+), 239 deletions(-) create mode 100644 crates/cellule-runtime/src/node/bundle/recovery.rs create mode 100644 crates/cellule-runtime/src/node/bundle/tests/recovery.rs rename crates/cellule-runtime/src/node/log/{recovery.rs => recovery/mod.rs} (50%) create mode 100644 crates/cellule-runtime/src/node/log/recovery/streaming.rs diff --git a/crates/cellule-ltx/docs/recovery.md b/crates/cellule-ltx/docs/recovery.md index 0572afd5..13eee871 100644 --- a/crates/cellule-ltx/docs/recovery.md +++ b/crates/cellule-ltx/docs/recovery.md @@ -28,3 +28,9 @@ still current; stale work cannot overwrite newer pages. A bucket listing is never a recovery selector. Runtime authority supplies the root, and reconstruction either matches it byte-for-byte or fails. + +`RecoveryOverlay::into_bundle_with_owner` transfers an immutable bundle together +with its original disk admission and artifact pin. Retain the returned +`BundleResourceOwner` until dispatched upload or cache work joins, including +after caller cancellation. This preserves unique file ownership for a cache +handoff. The ordinary `into_bundle` remains limited to unleased overlays. diff --git a/crates/cellule-ltx/src/replica/mod.rs b/crates/cellule-ltx/src/replica/mod.rs index af48e280..11374ceb 100644 --- a/crates/cellule-ltx/src/replica/mod.rs +++ b/crates/cellule-ltx/src/replica/mod.rs @@ -149,6 +149,15 @@ pub struct RecoveryOverlay { _bundle_lease: Option>, } +/// Original admission and artifact pin retained while a bundle transfers. +/// +/// Keep this owner through joined upload or cache work, including cancellation. +/// An owned bundle alone does not retain these caller-provided resources. +pub struct BundleResourceOwner { + _disk_reservation: Option, + _bundle_lease: Option>, +} + impl RecoveryOverlay { /// Describes the overlay that supersedes `predecessor`; the caller adds /// the disk reservation and bundle lease that keep it readable. @@ -193,6 +202,18 @@ impl RecoveryOverlay { Ok(self.bundle) } + /// Transfers the original immutable bundle and its resource owner together. + /// The receiver must retain the owner until its dispatched transfer joins. + pub fn into_bundle_with_owner(self) -> (crate::bundle::Bundle, BundleResourceOwner) { + ( + self.bundle, + BundleResourceOwner { + _disk_reservation: self._disk_reservation, + _bundle_lease: self._bundle_lease, + }, + ) + } + /// Returns the root this overlay supersedes. #[must_use] pub const fn predecessor(&self) -> RootRef { diff --git a/crates/cellule-runtime/docs/failover-and-followers.md b/crates/cellule-runtime/docs/failover-and-followers.md index 73a39739..2874a759 100644 --- a/crates/cellule-runtime/docs/failover-and-followers.md +++ b/crates/cellule-runtime/docs/failover-and-followers.md @@ -1565,6 +1565,24 @@ If `log.active` is true and no complete witness is available: ### Build recovery manifests +For the experimental bundle lane, recovery also inventories every authenticated +binding shard from the head retained by the fencing CAS. Follower scopes alone +omit Cells whose whole unmaterialized suffix is already selected in origin. +The coordinator verifies those selected prefixes, then feeds them and the +complete sealed follower witness through one file-backed builder. Overlapping +native sequences must have identical digests; missing Cells, command/physical +gaps and conflicting bytes reject before attachment. Scratch growth is admitted +before writing, and cache jobs retain that admission after caller cancellation. + +A bound Cell can pin only recovery from its original session and log epoch, +after canonical node fencing. Its binding remains pinned. The canonical node +seal refuses until every binding has a terminal materialized checkpoint; +uploading or attaching a manifest grants no transfer authority. Scheduling that +failed-boot materialization, interrupted enrollment resolution and terminal +catalog CAS is still unfinished, so ordinary bundle ACKs remain disabled. The +manifest reader admits up to 4,096 scopes under the existing 2 MiB byte ceiling; +the 2,000-scope codec test is format evidence, not node performance qualification. + The recoverer filters entries at or below the session's object-covered watermark, then groups the remaining verified frames by application, Cell, incarnation, and Cell epoch: diff --git a/crates/cellule-runtime/docs/write-performance-design.md b/crates/cellule-runtime/docs/write-performance-design.md index ab1e4f43..e69db42a 100644 --- a/crates/cellule-runtime/docs/write-performance-design.md +++ b/crates/cellule-runtime/docs/write-performance-design.md @@ -102,3 +102,11 @@ Implementation and prior measured results are tracked in the [bundle delivery record](../../../docs/bundle-coverage-implementation.md) and [WAL comparison](../../../docs/pr67-normal-wal-reevaluation.md). Keep bulk logs, raw request ledgers and source manifests outside the repository. + +The recovery coordinator now discovers selected-only Cells and joins exact +origin prefixes with the complete sealed follower witness in one admitted +file-backed reconstruction path. It rejects conflicting overlap and incomplete +Cell inventory. Bound overlays retain the original pin, and the canonical node +seal refuses unfinished materialization/checkpoint closure. Failed-boot root +materialization, terminal catalog selection and provisional enrollment resolution +remain required before enabling the canonical coverage frontier or actor ACKs. diff --git a/crates/cellule-runtime/src/control/authority/mod.rs b/crates/cellule-runtime/src/control/authority/mod.rs index 1befb383..d8764f3b 100644 --- a/crates/cellule-runtime/src/control/authority/mod.rs +++ b/crates/cellule-runtime/src/control/authority/mod.rs @@ -180,6 +180,14 @@ impl CellAuthority { if transition == Transition::BindBundle { crate::node::bundle::store::ensure_enrollment(&self.layout, &next).await?; } + if transition == Transition::AttachRecovery && observed.value.bundle_binding.is_some() { + crate::node::bundle::recovery::ensure_attachment( + &self.layout, + &observed.value, + next.recovery.as_ref().ok_or(Error::Fenced)?, + ) + .await?; + } if observed.value.bundle_binding.is_some() && observed.value.bundle_binding != next.bundle_binding { diff --git a/crates/cellule-runtime/src/control/mod.rs b/crates/cellule-runtime/src/control/mod.rs index e6ffc2a5..91dddcca 100644 --- a/crates/cellule-runtime/src/control/mod.rs +++ b/crates/cellule-runtime/src/control/mod.rs @@ -700,7 +700,10 @@ impl Control { || binding.session.as_bytes().iter().all(|byte| *byte == 0) || binding.digest.as_bytes().iter().all(|byte| *byte == 0) || self.state != ControlState::Serving - || self.recovery.is_some() + || self.recovery.as_ref().is_some_and(|recovery| { + recovery.leader_session != binding.session + || recovery.log_epoch != binding.epoch + }) || self .owner .as_ref() diff --git a/crates/cellule-runtime/src/control/tests.rs b/crates/cellule-runtime/src/control/tests.rs index 054029ec..ff4afec8 100644 --- a/crates/cellule-runtime/src/control/tests.rs +++ b/crates/cellule-runtime/src/control/tests.rs @@ -43,6 +43,40 @@ fn initial() -> Control { .unwrap() } +#[test] +fn bound_control_retains_only_its_original_recovery_lane() { + let mut control = initial(); + control.state = ControlState::Serving; + control.root = Some(root(7)); + control.bundle_binding = Some(BundleBindingRef { + session: owner(3).session, + epoch: 4, + digest: Digest::from_bytes([8; 32]), + }); + let attached = control.attach_recovery(recovery(root(7))).unwrap(); + assert_eq!( + Control::decode(&attached.encode().unwrap()).unwrap(), + attached + ); + for (session, epoch) in [(SessionId::from_bytes([5; 16]), 4), (owner(3).session, 5)] { + let mut foreign = recovery(root(7)); + foreign.leader_session = session; + foreign.log_epoch = epoch; + assert!(control.attach_recovery(foreign).is_err()); + } + let mut ordinary = attached.clone(); + ordinary.revision += 1; + ordinary.progress += 1; + ordinary.recovery = None; + ordinary.root = Some(root(9)); + assert!( + attached + .validate_transition(&ordinary, Transition::Publish) + .is_err(), + "ordinary publication cannot bypass pinned recovery" + ); +} + #[test] fn canonical_control_roundtrips_and_rejects_alternate_encodings() { let mut control = initial(); diff --git a/crates/cellule-runtime/src/node/advertisement/mod.rs b/crates/cellule-runtime/src/node/advertisement/mod.rs index 1ff5d656..04864f47 100644 --- a/crates/cellule-runtime/src/node/advertisement/mod.rs +++ b/crates/cellule-runtime/src/node/advertisement/mod.rs @@ -495,6 +495,7 @@ pub struct FencedNodeSession { pub(super) claim_generation: u64, pub(super) claim_expires_at_ms: i64, pub(super) log: Option, + pub(super) bundle: Option, } impl FencedNodeSession { @@ -534,6 +535,13 @@ impl FencedNodeSession { self.log.as_ref() } + /// Exact selected bundle head retained by the failed-session fencing CAS. + /// This observation grants no recovery completion or collection authority. + #[must_use] + pub const fn bundle_head(&self) -> Option { + self.bundle + } + /// Converts a fence into takeover authority when no fleet proof needs recovery. pub fn direct_takeover(&self) -> Result { if self.log.as_ref().is_some_and(NodeLogStatus::active) { diff --git a/crates/cellule-runtime/src/node/bundle/index/io.rs b/crates/cellule-runtime/src/node/bundle/index/io.rs index 9f9fab66..1a5da4d2 100644 --- a/crates/cellule-runtime/src/node/bundle/index/io.rs +++ b/crates/cellule-runtime/src/node/bundle/index/io.rs @@ -266,3 +266,38 @@ fn check_closed(bindings: &[Binding]) -> Result<()> { } Ok(()) } + +/// Stream authenticated binding rows without loading dense native histories. +/// Recovery must discover selected-only Cells as well as the follower scopes. +pub(in crate::node::bundle) async fn binding_inventory( + layout: &cellule_ltx::CellStorageLayout, + session: SessionId, + head: NodeBundleHead, +) -> Result> { + let Some(root) = load_root(layout, session, head).await? else { + return Ok( + super::super::store::load_legacy_catalog(layout, session, head) + .await? + .bindings, + ); + }; + let mut bindings = Vec::new(); + let mut pins = std::collections::HashSet::new(); + for (id, shard) in root.shards.iter().enumerate() { + let Some(shard) = shard else { continue }; + let (rows, _) = load_rows(layout, &root, shard, id as u8).await?; + for binding in rows { + let pin = binding + .control + .bundle_binding + .ok_or(Error::Node("bundle catalog lacks Cell pin"))?; + if !pins.insert(pin.digest) || pins.len() > MAX_BINDINGS { + return Err(Error::Node( + "bundle recovery inventory exceeds unique binding bound", + )); + } + bindings.push(binding); + } + } + Ok(bindings) +} diff --git a/crates/cellule-runtime/src/node/bundle/index/mod.rs b/crates/cellule-runtime/src/node/bundle/index/mod.rs index 53bc48dc..230c75f0 100644 --- a/crates/cellule-runtime/src/node/bundle/index/mod.rs +++ b/crates/cellule-runtime/src/node/bundle/index/mod.rs @@ -27,7 +27,7 @@ pub(super) fn history_extent(catalog: &Catalog, pin: Digest) -> Option .get(pin.as_bytes()) .map(|history| history.extent.clone()) } -pub(super) use io::{ensure_drained, load_cells}; +pub(super) use io::{binding_inventory, ensure_drained, load_cells}; pub(super) const HEADER_BYTES: usize = 32 << 10; const SHARDS: usize = 256; diff --git a/crates/cellule-runtime/src/node/bundle/mod.rs b/crates/cellule-runtime/src/node/bundle/mod.rs index 1a81bd5e..a937b479 100644 --- a/crates/cellule-runtime/src/node/bundle/mod.rs +++ b/crates/cellule-runtime/src/node/bundle/mod.rs @@ -40,6 +40,7 @@ mod closure; mod codec; mod index; mod proof; +pub(crate) mod recovery; mod selection; #[cfg(test)] use proof::checkpoint_prefix; diff --git a/crates/cellule-runtime/src/node/bundle/proof.rs b/crates/cellule-runtime/src/node/bundle/proof.rs index 3d8b9f3b..3a900a12 100644 --- a/crates/cellule-runtime/src/node/bundle/proof.rs +++ b/crates/cellule-runtime/src/node/bundle/proof.rs @@ -41,35 +41,58 @@ impl NodeDirectory { return Err(Error::Fenced); } let head = head.ok_or(Error::PendingPublication)?; - let cells = [(*authority.layout().application_id(), *value.cell.as_bytes())] - .into_iter() - .collect(); - let mut catalog = store::load_catalog_cells(&self.layout, session, head, &cells).await?; - let binding = catalog.binding_mut(pin.digest)?.clone(); - if binding.phase == BindingPhase::Provisional { - return Err(Error::PendingPublication); - } - if head.epoch != pin.epoch - || binding.control.bundle_binding != Some(pin) - || binding.application.as_bytes() != authority.layout().application_id() - || binding.control.cell != value.cell - || binding.control.incarnation != value.incarnation - || binding.control.epoch != value.epoch - || binding.control.code != value.code - || binding.control.schema != value.schema - { - return Err(Error::Fenced); - } - let frames = verify_binding(&self.layout, session, head.epoch, &binding, limits).await?; - let current = value.ltx_root().ok_or(Error::Fenced)?; - checkpoint_prefix(&binding, &frames, ¤t)?; - Ok(BundleCoverageProof { + load_coverage_at(&self.layout, head, authority, control, limits) + .await + .map(|(proof, _)| proof) + } +} + +pub(super) async fn load_coverage_at( + layout: &cellule_ltx::CellStorageLayout, + head: NodeBundleHead, + authority: &CellAuthority, + control: &VersionedControl, + limits: cellule_ltx::Limits, +) -> Result<(BundleCoverageProof, Vec)> { + let value = control.value(); + let pin = value.bundle_binding.ok_or(Error::PendingPublication)?; + if authority.layout().node_path(pin.session.as_bytes()) + != layout.node_path(pin.session.as_bytes()) + || authority.layout().immutable_cache_identity() != layout.immutable_cache_identity() + { + return Err(Error::Fenced); + } + let cells = [(*authority.layout().application_id(), *value.cell.as_bytes())] + .into_iter() + .collect(); + let mut catalog = store::load_catalog_cells(layout, pin.session, head, &cells).await?; + let binding = catalog.binding_mut(pin.digest)?.clone(); + if binding.phase == BindingPhase::Provisional { + return Err(Error::PendingPublication); + } + if head.epoch != pin.epoch + || binding.control.bundle_binding != Some(pin) + || binding.application.as_bytes() != authority.layout().application_id() + || binding.control.cell != value.cell + || binding.control.incarnation != value.incarnation + || binding.control.epoch != value.epoch + || binding.control.code != value.code + || binding.control.schema != value.schema + { + return Err(Error::Fenced); + } + let frames = verify_binding(layout, pin.session, head.epoch, &binding, limits).await?; + let current = value.ltx_root().ok_or(Error::Fenced)?; + checkpoint_prefix(&binding, &frames, ¤t)?; + Ok(( + BundleCoverageProof { pin, binding, head, - session, - }) - } + session: pin.session, + }, + frames, + )) } pub(super) async fn verify_binding( diff --git a/crates/cellule-runtime/src/node/bundle/recovery.rs b/crates/cellule-runtime/src/node/bundle/recovery.rs new file mode 100644 index 00000000..5f47da7c --- /dev/null +++ b/crates/cellule-runtime/src/node/bundle/recovery.rs @@ -0,0 +1,142 @@ +//! Failed-session inventory and exact selected-prefix reconstruction. +use super::*; +use crate::control::authority::{CellAuthority, VersionedControl}; +use crate::node::FencedNodeSession; +use crate::node::directory::NodeRecord; + +async fn record(layout: &cellule_ltx::CellStorageLayout, session: SessionId) -> Result { + let (bytes, _) = layout + .store() + .get_with_etag_bounded( + &layout.node_path(session.as_bytes()), + crate::node::MAX_NODE_BYTES, + ) + .await?; + let record = NodeRecord::decode_canonical(&bytes)?; + let actual_session = match &record { + NodeRecord::Advertisement(node) => node.session, + NodeRecord::Tombstone(node) => node.session, + }; + if actual_session != session { + return Err(Error::Fenced); + } + Ok(record) +} + +/// Discovery remains advisory; only a fenced claim can attach recovered state. +/// Authenticate every shard without loading dense histories during discovery. +pub(crate) async fn inventory_for_owner( + layout: &cellule_ltx::CellStorageLayout, + session: SessionId, +) -> Result> { + let node = match record(layout, session).await { + Ok(node) => node, + // Scope-only catalog discovery also supports applications that do not + // use a node directory. A bound Cell still fails the fenced lookup. + Err(Error::Storage(cellule_store::StorageError::NotFound { .. })) => return Ok(Vec::new()), + Err(source) => return Err(source), + }; + let head = match node { + NodeRecord::Advertisement(node) => node.bundle, + NodeRecord::Tombstone(node) => node.bundle, + }; + let Some(head) = head else { + return Ok(Vec::new()); + }; + inventory_at(layout, session, head).await +} + +async fn inventory_at( + layout: &cellule_ltx::CellStorageLayout, + session: SessionId, + head: NodeBundleHead, +) -> Result> { + Ok(index::binding_inventory(layout, session, head) + .await? + .into_iter() + .filter(|binding| binding.phase != BindingPhase::Closed) + .map(|binding| (binding.application, binding.control)) + .collect()) +} + +pub(crate) async fn fenced_inventory( + layout: &cellule_ltx::CellStorageLayout, + fenced: &FencedNodeSession, +) -> Result> { + let Some(head) = fenced.bundle_head() else { + return Ok(Vec::new()); + }; + let NodeRecord::Tombstone(current) = record(layout, fenced.session()).await? else { + return Err(Error::Fenced); + }; + if current.node != fenced.node() + || current.claimant != Some(fenced.claimant()) + || current.claim_generation != fenced.claim_generation() + || current.claim_expires_at_ms != Some(fenced.claim_expires_at_ms()) + || current.log.as_ref() != fenced.log() + || current.bundle != Some(head) + { + return Err(Error::Fenced); + } + inventory_at(layout, fenced.session(), head).await +} + +pub(crate) async fn selected_frames( + layout: &cellule_ltx::CellStorageLayout, + authority: &CellAuthority, + control: &VersionedControl, + fenced: &FencedNodeSession, + limits: cellule_ltx::Limits, +) -> Result> { + let pin = control.value().bundle_binding.ok_or(Error::Fenced)?; + let head = fenced.bundle_head().ok_or(Error::Fenced)?; + if pin.session != fenced.session() || pin.epoch != head.epoch { + return Err(Error::Fenced); + } + // Return original frames, including a materialized overlap, so the global + // follower witness must agree byte-for-byte with selected native sequence. + super::proof::load_coverage_at(layout, head, authority, control, limits) + .await + .map(|(_, frames)| frames) +} + +/// A bound writer may retain a recovery overlay only after canonical fencing. +/// Keep the original pin until the complete suffix has a materialized root; +/// neither this attachment nor a manifest upload grants departure authority. +pub(crate) async fn ensure_attachment( + layout: &cellule_ltx::CellStorageLayout, + control: &Control, + recovery: &crate::control::RecoveryOverlayRef, +) -> Result<()> { + let pin = control.bundle_binding.ok_or(Error::Fenced)?; + let NodeRecord::Tombstone(node) = record(layout, pin.session).await? else { + return Err(Error::Fenced); + }; + if node.claimant.is_none() + || node.log.as_ref().is_none_or(|log| { + log.phase() != crate::node::log_state::NodeLogPhase::Recovering + || log.epoch() != pin.epoch + }) + || recovery.leader_session != pin.session + || recovery.log_epoch != pin.epoch + { + return Err(Error::Fenced); + } + let head = node.bundle.ok_or(Error::PendingPublication)?; + let cells = [(*layout.application_id(), *control.cell.as_bytes())] + .into_iter() + .collect(); + let mut catalog = store::load_catalog_cells(layout, pin.session, head, &cells).await?; + let binding = catalog.binding_mut(pin.digest)?; + if !matches!(binding.phase, BindingPhase::Open | BindingPhase::Closing) + || binding.control.incarnation != control.incarnation + || binding.control.epoch != control.epoch + || binding.control.code != control.code + || binding.control.schema != control.schema + || recovery.final_commit_sequence < binding.selected_commit + || recovery.final_txid < binding.selected_position.txid + { + return Err(Error::PendingPublication); + } + Ok(()) +} diff --git a/crates/cellule-runtime/src/node/bundle/tests/mod.rs b/crates/cellule-runtime/src/node/bundle/tests/mod.rs index 2848dcd5..353a42b7 100644 --- a/crates/cellule-runtime/src/node/bundle/tests/mod.rs +++ b/crates/cellule-runtime/src/node/bundle/tests/mod.rs @@ -19,6 +19,7 @@ mod faults; mod index; mod lifecycle; mod ranges; +mod recovery; struct Fixture { count: Arc, layout: CellStorageLayout, diff --git a/crates/cellule-runtime/src/node/bundle/tests/recovery.rs b/crates/cellule-runtime/src/node/bundle/tests/recovery.rs new file mode 100644 index 00000000..66bce130 --- /dev/null +++ b/crates/cellule-runtime/src/node/bundle/tests/recovery.rs @@ -0,0 +1,396 @@ +use super::*; +use crate::follower::{FollowerReceipt, FollowerStore}; +use crate::node::log_recovery::{NodeLogRecovery, RecoveryCell, RecoveryCoordinator}; +use crate::node::log_state::NodeLogStatus; +use crate::node::log_transport::{ + AppendRequest, NodeLogTransport, RetireRequest, SealRequest, TailRequest, +}; +use crate::recovery::manifest::RecoveryManifestStore; +use futures_util::future::BoxFuture; + +struct Followers(Vec<(NodeId, FollowerStore)>); +impl Followers { + fn get(&self, member: NodeId) -> Result<&FollowerStore> { + self.0 + .iter() + .find_map(|(id, store)| (*id == member).then_some(store)) + .ok_or(Error::Node("test follower absent")) + } +} +impl NodeLogTransport for Followers { + fn append<'a>( + &'a self, + member: NodeId, + request: AppendRequest, + ) -> BoxFuture<'a, Result> { + Box::pin(async move { + self.get(member)? + .append( + request.leader_session, + request.log_epoch, + request.frames, + request.covered_through, + ) + .await + }) + } + fn seal<'a>( + &'a self, + member: NodeId, + request: SealRequest, + ) -> BoxFuture<'a, Result> { + Box::pin(async move { + self.get(member)? + .seal(request.leader_session, request.log_epoch) + .await + }) + } + fn tail<'a>( + &'a self, + member: NodeId, + request: TailRequest, + ) -> BoxFuture<'a, Result>> { + Box::pin(async move { + self.get(member)? + .read_tail( + request.leader_session, + request.log_epoch, + request.first_sequence, + ) + .await + }) + } + fn retire<'a>( + &'a self, + member: NodeId, + request: RetireRequest, + ) -> BoxFuture<'a, Result> { + Box::pin(async move { + self.get(member)? + .retire( + request.leader_session, + request.log_epoch, + request.covered_through, + ) + .await + }) + } +} + +async fn enroll(f: &mut Fixture) { + let mut node = f.node.advertisement().clone(); + node.log = Some( + NodeLogStatus::open( + node.node, + EPOCH, + vec![NodeId::from_bytes([2; 16]), NodeId::from_bytes([3; 16])], + ) + .unwrap() + .activate(node.node) + .unwrap(), + ); + node.generation += 1; + f.node = f + .directory + .update_advertisement(&f.node, node, NOW) + .await + .unwrap(); +} + +async fn fence(f: &Fixture) -> crate::node::FencedNodeSession { + let now = NOW + 31_000; + let mut claimant = f.node.advertisement().clone(); + claimant.node = NodeId::from_bytes([8; 16]); + claimant.session = SessionId::from_bytes([8; 16]); + claimant.log = None; + claimant.bundle = None; + claimant.issued_at_ms = now; + claimant.expires_at_ms = now + 30_000; + claimant.signature = SigningKey::from_bytes(&[10; 32]) + .sign(&claimant.signing_bytes().unwrap()) + .to_bytes(); + f.directory.create(claimant, now).await.unwrap(); + f.directory + .claim_expired( + SessionId::from_bytes([1; 16]), + SessionId::from_bytes([8; 16]), + now, + ) + .await + .unwrap() +} + +enum Scenario { + Pruned, + Overlap, + SelectedOnly, + Conflict, + MissingCell, +} + +#[tokio::test] +async fn recovery_joins_selected_only_cell_and_pruned_prefix_with_prior_fleet_ack() { + verify_recovery(Scenario::Pruned).await; +} +#[tokio::test] +async fn recovery_checks_identical_follower_overlap_against_selected_prefix() { + verify_recovery(Scenario::Overlap).await; +} +#[tokio::test] +async fn recovery_restores_selected_cells_with_an_empty_follower_witness() { + verify_recovery(Scenario::SelectedOnly).await; +} +#[tokio::test] +async fn recovery_rejects_conflicting_follower_overlap_without_attachment() { + verify_recovery(Scenario::Conflict).await; +} +#[tokio::test] +async fn recovery_rejects_omitted_selected_only_cell_without_attachment() { + verify_recovery(Scenario::MissingCell).await; +} + +async fn verify_recovery(scenario: Scenario) { + let mut f = Fixture::new().await; + enroll(&mut f).await; + let mut a = f.cell(4).await; + let mut b = f.cell(5).await; + let (_, mut prefix, arange) = f.append(&mut a, 2); + let (_, bframes, brange) = f.append(&mut b, 2); + prefix.extend(bframes); + let proposal = f + .directory + .prepare_node_bundle(&f.node, &prefix, &[arange, brange], NOW) + .await + .unwrap(); + let (selected, _) = f + .directory + .select_node_bundle(&f.node, &proposal, &f.lease, Limits::default(), NOW) + .await + .unwrap(); + let selected_through = selected + .advertisement() + .bundle_head() + .unwrap() + .selected_through(); + // Exercise the state that bundle ACK integration must support: the shared + // selected prefix is retained in origin while Cell roots still lag. + let coverage = if matches!(scenario, Scenario::Overlap | Scenario::Conflict) { + 0 + } else { + selected_through + }; + f.node = f + .directory + .advance_log_coverage(&selected, coverage, NOW) + .await + .unwrap(); + let (suffix, fleet) = if matches!(scenario, Scenario::SelectedOnly) { + (Vec::new(), None) + } else { + let (_, suffix, fleet) = f.append(&mut a, 3); + (suffix, Some(fleet)) + }; + let mut transmitted = prefix + .iter() + .map(|frame| frame.encoded().clone()) + .collect::>(); + if matches!(scenario, Scenario::Conflict) { + let first = &prefix[0]; + let mut scope = first.scope(); + scope.cell_epoch += 1; + transmitted[0] = cellule_ltx::encode_node_frame( + scope, + first.segment().clone(), + first.body().clone(), + Limits::default(), + ) + .unwrap() + .encoded() + .clone(); + } + let dirs = [tempfile::tempdir().unwrap(), tempfile::tempdir().unwrap()]; + let followers = Arc::new(Followers( + dirs.iter() + .enumerate() + .map(|(i, dir)| { + ( + NodeId::from_bytes([i as u8 + 2; 16]), + FollowerStore::open( + dir.path().to_owned(), + Limits::default(), + cellule_ltx::DiskBudget::new(1 << 30), + ) + .unwrap(), + ) + }) + .collect(), + )); + for member in [2, 3] { + let id = NodeId::from_bytes([member; 16]); + let receipt = followers + .append( + id, + AppendRequest { + leader_session: SessionId::from_bytes([1; 16]), + log_epoch: EPOCH, + frames: transmitted.clone(), + covered_through: 0, + }, + ) + .await + .unwrap(); + let receipt = if suffix.is_empty() { + receipt + } else { + followers + .append( + id, + AppendRequest { + leader_session: SessionId::from_bytes([1; 16]), + log_epoch: EPOCH, + frames: suffix.iter().map(|frame| frame.encoded().clone()).collect(), + covered_through: coverage, + }, + ) + .await + .unwrap() + }; + if coverage > 0 && !suffix.is_empty() { + assert!(receipt.base_sequence > selected_through); + } + f.gate.acknowledge(id, receipt.durable_through).unwrap(); + } + f.gate.activate_fleet().unwrap(); + if let Some(fleet) = fleet { + assert_eq!( + f.gate.prove(fleet.ticket()).await.unwrap().source(), + DurabilitySource::Fleet + ); + } + let fenced = fence(&f).await; + f.lease.fence(); + let transport: Arc = followers; + let recovery = NodeLogRecovery::from_fenced(transport, &fenced, Limits::default()).unwrap(); + let sealed = recovery.ensure_sealed_bounded().await.unwrap(); + let expected_frames = if coverage == 0 { + prefix.len() + suffix.len() + } else { + suffix.len() + }; + assert_eq!(sealed.frame_count(), expected_frames as u64); + if matches!(scenario, Scenario::Pruned) { + assert_eq!( + sealed.scopes(Limits::default()).unwrap().len(), + 1, + "the selected-only Cell is absent from follower scopes" + ); + } + let inventory = super::super::recovery::inventory_for_owner(&f.layout, fenced.session()) + .await + .unwrap(); + assert_eq!( + inventory.len(), + 2, + "complete binding discovery includes selected-only Cells" + ); + let manifests = RecoveryManifestStore::new(f.layout.clone(), Limits::default()) + .with_recovery_scratch(f.scratch.path().to_owned()); + let cells = [&a, &b] + .into_iter() + .filter(|cell| { + !matches!(scenario, Scenario::MissingCell) + || cell.control.value().cell != b.control.value().cell + }) + .map(|cell| RecoveryCell { + application: ApplicationId::from_bytes(*cell.authority.layout().application_id()), + authority: cell.authority.clone(), + observed: cell.control.clone(), + }) + .collect(); + let coordinator = RecoveryCoordinator::new(recovery, manifests.clone()); + let result = coordinator + .recover_sealed(fenced.clone(), cells, sealed) + .await; + if matches!(scenario, Scenario::Conflict | Scenario::MissingCell) { + let expected = if matches!(scenario, Scenario::Conflict) { + "follower witness conflicts with selected bundle prefix" + } else { + "selected bundle Cell is absent from recovery inventory" + }; + assert!(matches!(result, Err(Error::Node(message)) if message == expected)); + for cell in [&a, &b] { + assert!( + cell.authority + .load(cell.control.value().cell) + .await + .unwrap() + .unwrap() + .value() + .recovery + .is_none() + ); + } + return; + } + let controls = result.unwrap(); + assert_eq!( + controls.len(), + 2, + "selected-only Cells also require durable recovery" + ); + for cell in [&a, &b] { + let control = controls + .iter() + .find(|control| control.value().cell == cell.control.value().cell) + .unwrap(); + let overlay = manifests + .load_overlay( + cell.control.value().cell, + cell.control.value().incarnation, + control.value().recovery.as_ref().unwrap(), + ) + .await + .unwrap(); + let prepared = cell + .replica + .prepare_recovered_overlay(&overlay, 1) + .await + .unwrap(); + let path = f.scratch.path().join(format!( + "recovered-{}.sqlite", + cell.control.value().cell.as_bytes()[0] + )); + cell.replica + .open_root(&prepared.root()) + .await + .unwrap() + .restore(&path) + .await + .unwrap(); + let db = cellule_ltx::rusqlite::Connection::open(path).unwrap(); + let last = if cell.control.value().cell == a.control.value().cell && fleet.is_some() { + 3 + } else { + 2 + }; + for command in 2..=last { + let result: String = db + .query_row( + "SELECT result FROM outcomes WHERE request=?1", + [format!("request-{command}")], + |row| row.get(0), + ) + .unwrap(); + assert_eq!(result, format!("result-{command}")); + } + } + assert!( + matches!( + coordinator + .finish(&f.directory, fenced, controls, NOW + 31_000) + .await, + Err(Error::PendingPublication) + ), + "pinned recovery alone cannot close bound Cells or admit transfer" + ); +} diff --git a/crates/cellule-runtime/src/node/directory/mod.rs b/crates/cellule-runtime/src/node/directory/mod.rs index 0f893538..f13e8ede 100644 --- a/crates/cellule-runtime/src/node/directory/mod.rs +++ b/crates/cellule-runtime/src/node/directory/mod.rs @@ -648,6 +648,7 @@ impl NodeTombstone { .claim_expires_at_ms .ok_or(Error::Node("node recovery claim expiry is missing"))?, log: self.log.clone(), + bundle: self.bundle, }) } diff --git a/crates/cellule-runtime/src/node/directory/recovery.rs b/crates/cellule-runtime/src/node/directory/recovery.rs index a5304c2c..c642dbf2 100644 --- a/crates/cellule-runtime/src/node/directory/recovery.rs +++ b/crates/cellule-runtime/src/node/directory/recovery.rs @@ -520,6 +520,15 @@ impl NodeDirectory { else { return Err(Error::Fenced); }; + // A recovery manifest is not terminal bundle closure. Keep the whole + // failed boot fenced until every original binding is materialized and + // checkpointed; lower-level callers cannot bypass the coordinator. + crate::node::bundle::store::ensure_session_drained( + &self.layout, + current.session, + current.bundle, + ) + .await?; if let Some(log) = ¤t.log && matches!(log.phase(), NodeLogPhase::Sealed | NodeLogPhase::Retired) && log.recovery_manifest() == recovery_manifest diff --git a/crates/cellule-runtime/src/node/log/recovery.rs b/crates/cellule-runtime/src/node/log/recovery/mod.rs similarity index 50% rename from crates/cellule-runtime/src/node/log/recovery.rs rename to crates/cellule-runtime/src/node/log/recovery/mod.rs index 73279ca0..ff2f3e06 100644 --- a/crates/cellule-runtime/src/node/log/recovery.rs +++ b/crates/cellule-runtime/src/node/log/recovery/mod.rs @@ -6,6 +6,12 @@ use super::*; +mod streaming; +pub(crate) use streaming::StreamingRecovery; +pub use streaming::{ + build_recovery_overlays_file_backed, build_recovery_overlays_file_backed_stream, +}; + /// Exact published Cell state used to validate a recovered node-log witness. #[derive(Clone, Copy, Debug, PartialEq, Eq)] pub struct RecoveryBase { @@ -163,183 +169,6 @@ pub fn build_recovery_overlays( Ok(recovered) } -/// Splits a complete witness into per-Cell overlays without retaining segment -/// bodies in the heap. Each Cell gets one scratch-backed bundle builder and -/// every frame body is released after its row is appended. -pub fn build_recovery_overlays_file_backed( - frames: Vec, - bases: &[RecoveryBase], - limits: cellule_ltx::Limits, - scratch: &Path, -) -> Result> { - build_recovery_overlays_file_backed_stream(frames.into_iter().map(Ok), bases, limits, scratch) -} - -/// Streaming variant used by the bounded witness reader. The iterator may -/// yield one verified frame at a time; no complete witness is retained. -pub fn build_recovery_overlays_file_backed_stream( - frames: I, - bases: &[RecoveryBase], - limits: cellule_ltx::Limits, - scratch: &Path, -) -> Result> -where - I: IntoIterator>, -{ - let mut frames = frames.into_iter(); - let first = frames - .next() - .ok_or(Error::Node("recovery witness is empty"))??; - let leader = first.scope().leader_session; - let log_epoch = first.scope().log_epoch; - let frames = std::iter::once(Ok(first)).chain(frames); - - type Key = ([u8; 16], [u8; 32], [u8; 16], u64); - let base_by_key = bases - .iter() - .map(|base| { - ( - ( - base.application, - base.root.cell, - base.root.incarnation, - base.cell_epoch, - ), - *base, - ) - }) - .collect::>(); - if base_by_key.len() != bases.len() { - return Err(Error::Node("recovery bases contain duplicate Cell scope")); - } - - struct CellBuilder { - base: RecoveryBase, - bundle: cellule_ltx::bundle::BundleBuilder, - first_node_sequence: u64, - last_node_sequence: u64, - last_commit_sequence: u64, - last_first_commit_sequence: u64, - final_position: cellule_ltx::Position, - } - - let mut previous_node_sequence = None::; - let mut grouped = BTreeMap::::new(); - for frame in frames { - let frame = frame?; - let scope = frame.scope(); - if scope.leader_session != leader || scope.log_epoch != log_epoch { - return Err(Error::Node("recovery witness has mixed sessions")); - } - if previous_node_sequence - .is_some_and(|previous| previous.checked_add(1) != Some(scope.node_sequence)) - { - return Err(Error::Node("recovery witness is not contiguous")); - } - previous_node_sequence = Some(scope.node_sequence); - let key = ( - scope.application, - scope.cell, - scope.incarnation, - scope.cell_epoch, - ); - let base = base_by_key - .get(&key) - .copied() - .ok_or(Error::Node("recovery frame has no exact published base"))?; - let covered = scope.commit_sequence <= base.root.commit_sequence; - if !covered && frame.first_commit_sequence() <= base.root.commit_sequence { - return Err(Error::Node( - "published base splits a node-log command group", - )); - } - let position = frame.segment().position(); - if covered != (position.txid <= base.root.position.txid) - || (position.txid == base.root.position.txid && position != base.root.position) - { - return Err(Error::Node("recovery frame disagrees with published base")); - } - if covered { - if grouped.contains_key(&key) { - return Err(Error::Node( - "recovery Cell covered prefix follows uncovered tail", - )); - } - continue; - } - - if let Some(cell) = grouped.get_mut(&key) { - if !range_continues( - cell.last_first_commit_sequence, - cell.last_commit_sequence, - frame.first_commit_sequence(), - scope.commit_sequence, - ) { - return Err(Error::Node("recovery Cell commit sequence has a gap")); - } - cell.bundle - .push(cellule_ltx::bundle::BundleEntry::for_cell( - base.root.cell, - base.root.incarnation, - frame.segment().clone(), - frame.body().to_vec(), - ))?; - cell.last_node_sequence = scope.node_sequence; - cell.last_commit_sequence = scope.commit_sequence; - cell.last_first_commit_sequence = frame.first_commit_sequence(); - cell.final_position = position; - continue; - } - - let first_commit = base - .root - .commit_sequence - .checked_add(1) - .ok_or(Error::Node("recovery commit sequence overflow"))?; - if frame.first_commit_sequence() != first_commit { - return Err(Error::Node("recovery Cell commit sequence has a gap")); - } - let mut bundle = cellule_ltx::bundle::BundleBuilder::new_temp(scratch, limits)?; - bundle.push(cellule_ltx::bundle::BundleEntry::for_cell( - base.root.cell, - base.root.incarnation, - frame.segment().clone(), - frame.body().to_vec(), - ))?; - grouped.insert( - key, - CellBuilder { - base, - bundle, - first_node_sequence: scope.node_sequence, - last_node_sequence: scope.node_sequence, - last_commit_sequence: scope.commit_sequence, - last_first_commit_sequence: frame.first_commit_sequence(), - final_position: position, - }, - ); - } - - grouped - .into_values() - .map(|cell| { - let bundle = cell.bundle.finish()?; - Ok(RecoveredCellTail { - application: cell.base.application, - cell_epoch: cell.base.cell_epoch, - first_node_sequence: cell.first_node_sequence, - last_node_sequence: cell.last_node_sequence, - overlay: cellule_ltx::RecoveryOverlay::new( - cell.base.root, - bundle, - cell.final_position, - cell.last_commit_sequence, - ), - }) - }) - .collect() -} - fn range_continues(first: u64, last: u64, next_first: u64, next_last: u64) -> bool { // Checkpoint cuts may repeat one complete range. A later transaction must // start immediately after it; sharing only an endpoint cannot hide a gap. diff --git a/crates/cellule-runtime/src/node/log/recovery/streaming.rs b/crates/cellule-runtime/src/node/log/recovery/streaming.rs new file mode 100644 index 00000000..ec0eb6d9 --- /dev/null +++ b/crates/cellule-runtime/src/node/log/recovery/streaming.rs @@ -0,0 +1,301 @@ +//! One bounded file-backed reconstruction path for selected and follower tails. +use super::*; + +type Key = ([u8; 16], [u8; 32], [u8; 16], u64); +const MAX_BASES: usize = 4_096; +const MAX_PREFIX_FRAMES: usize = 256; + +struct CellBuilder { + base: RecoveryBase, + bundle: cellule_ltx::bundle::BundleBuilder, + reservation: Option, + first_node_sequence: u64, + last_node_sequence: u64, + last_commit_sequence: u64, + last_first_commit_sequence: u64, + final_position: cellule_ltx::Position, + seeded: bool, +} + +pub(crate) struct StreamingRecovery { + bases: BTreeMap, + grouped: BTreeMap, + // The selected prefix is bounded by 4096 bindings * 256 frames. Exact + // digest comparison prevents an overlapping follower row from replacing it. + prefix_digests: BTreeMap, + leader: [u8; 16], + epoch: u64, + previous: Option, + limits: cellule_ltx::Limits, + scratch: std::path::PathBuf, + disk: Option, +} + +impl StreamingRecovery { + pub(crate) fn new( + bases: &[RecoveryBase], + leader: [u8; 16], + epoch: u64, + limits: cellule_ltx::Limits, + scratch: &Path, + disk: Option, + ) -> Result { + if bases.len() > MAX_BASES { + return Err(Error::Capacity("recovery Cell bases")); + } + let map = bases + .iter() + .map(|base| { + ( + ( + base.application, + base.root.cell, + base.root.incarnation, + base.cell_epoch, + ), + *base, + ) + }) + .collect::>(); + if map.len() != bases.len() { + return Err(Error::Node("recovery bases contain duplicate Cell scope")); + } + Ok(Self { + bases: map, + grouped: BTreeMap::new(), + prefix_digests: BTreeMap::new(), + leader, + epoch, + previous: None, + limits, + scratch: scratch.to_owned(), + disk, + }) + } + + /// Seeds one dependency-verified complete Cell prefix. Native rows can be + /// interleaved across Cells, so only each Cell's sequence must increase here. + pub(crate) fn seed(&mut self, frames: Vec) -> Result<()> { + if frames.len() > MAX_PREFIX_FRAMES { + return Err(Error::Capacity("recovery selected prefix frames")); + } + for frame in frames { + let sequence = frame.scope().node_sequence; + if self.prefix_digests.len() >= MAX_BASES * MAX_PREFIX_FRAMES + || self + .prefix_digests + .insert(sequence, frame.digest()) + .is_some() + { + return Err(Error::Node( + "recovery selected prefix repeats native sequence", + )); + } + self.append(frame, true)?; + } + Ok(()) + } + + /// Verifies the entire contiguous follower witness, including any overlap + /// with selected origin. Only byte-identical overlap can be skipped. + pub(crate) fn push(&mut self, frame: cellule_ltx::VerifiedNodeFrame) -> Result<()> { + let scope = frame.scope(); + if scope.leader_session != self.leader || scope.log_epoch != self.epoch { + return Err(Error::Node("recovery witness has mixed sessions")); + } + if self + .previous + .is_some_and(|before| before.checked_add(1) != Some(scope.node_sequence)) + { + return Err(Error::Node("recovery witness is not contiguous")); + } + self.previous = Some(scope.node_sequence); + if let Some(digest) = self.prefix_digests.get(&scope.node_sequence) { + if *digest != frame.digest() { + return Err(Error::Node( + "follower witness conflicts with selected bundle prefix", + )); + } + return Ok(()); + } + self.append(frame, false) + } + + fn append(&mut self, frame: cellule_ltx::VerifiedNodeFrame, seeded: bool) -> Result<()> { + let scope = frame.scope(); + if scope.leader_session != self.leader || scope.log_epoch != self.epoch { + return Err(Error::Node("recovery witness has mixed sessions")); + } + let key = ( + scope.application, + scope.cell, + scope.incarnation, + scope.cell_epoch, + ); + let base = self + .bases + .get(&key) + .copied() + .ok_or(Error::Node("recovery frame has no exact published base"))?; + let covered = scope.commit_sequence <= base.root.commit_sequence; + if !covered && frame.first_commit_sequence() <= base.root.commit_sequence { + return Err(Error::Node( + "published base splits a node-log command group", + )); + } + let position = frame.segment().position(); + if covered != (position.txid <= base.root.position.txid) + || (position.txid == base.root.position.txid && position != base.root.position) + { + return Err(Error::Node("recovery frame disagrees with published base")); + } + if covered { + if self + .grouped + .get(&key) + .is_some_and(|cell| !cell.seeded || seeded) + { + return Err(Error::Node( + "recovery Cell covered prefix follows uncovered tail", + )); + } + return Ok(()); + } + if !self.grouped.contains_key(&key) { + if base.root.commit_sequence.checked_add(1) != Some(frame.first_commit_sequence()) { + return Err(Error::Node("recovery Cell commit sequence has a gap")); + } + let reservation = self + .disk + .as_ref() + .map(|disk| disk.try_reserve(64)) + .transpose()?; + self.grouped.insert( + key, + CellBuilder { + base, + bundle: cellule_ltx::bundle::BundleBuilder::new_temp( + &self.scratch, + self.limits, + )?, + reservation, + first_node_sequence: scope.node_sequence, + last_node_sequence: 0, + last_commit_sequence: base.root.commit_sequence, + last_first_commit_sequence: base.root.commit_sequence, + final_position: base.root.position, + seeded, + }, + ); + } + let cell = self + .grouped + .get_mut(&key) + .ok_or(Error::Node("recovery builder disappeared"))?; + if scope.node_sequence <= cell.last_node_sequence + || !range_continues( + cell.last_first_commit_sequence, + cell.last_commit_sequence, + frame.first_commit_sequence(), + scope.commit_sequence, + ) + || cell.final_position.txid.checked_add(1) != Some(frame.segment().min_txid) + || cell.final_position.checksum != frame.segment().pre_checksum + { + return Err(Error::Node( + "recovery Cell commit or physical range has a gap", + )); + } + if let Some(reservation) = &cell.reservation { + // Native envelopes plus 256 bytes per row conservatively admit the + // bundle's encoded footer before any scratch-file growth. + reservation.try_grow( + (frame.encoded().len() as u64) + .checked_add(256) + .ok_or(Error::Capacity("recovery scratch row bytes"))?, + )?; + } + cell.bundle + .push(cellule_ltx::bundle::BundleEntry::for_cell( + base.root.cell, + base.root.incarnation, + frame.segment().clone(), + frame.body().to_vec(), + ))?; + cell.last_node_sequence = scope.node_sequence; + cell.last_commit_sequence = scope.commit_sequence; + cell.last_first_commit_sequence = frame.first_commit_sequence(); + cell.final_position = position; + cell.seeded |= seeded; + Ok(()) + } + + pub(crate) fn finish(self) -> Result> { + self.grouped + .into_values() + .map(|cell| { + let bundle = cell.bundle.finish()?; + let mut overlay = cellule_ltx::RecoveryOverlay::new( + cell.base.root, + bundle, + cell.final_position, + cell.last_commit_sequence, + ); + if let Some(reservation) = cell.reservation { + if overlay.bundle().len() > reservation.bytes() { + return Err(Error::Capacity("recovery scratch exceeds admission")); + } + reservation.resize(overlay.bundle().len())?; + overlay = overlay.with_disk_reservation(reservation); + } + Ok(RecoveredCellTail { + application: cell.base.application, + cell_epoch: cell.base.cell_epoch, + first_node_sequence: cell.first_node_sequence, + last_node_sequence: cell.last_node_sequence, + overlay, + }) + }) + .collect() + } +} + +/// Splits a complete witness into file-backed overlays, releasing every frame +/// body after its row is appended. Bases remain exact authority-pinned roots. +pub fn build_recovery_overlays_file_backed( + frames: Vec, + bases: &[RecoveryBase], + limits: cellule_ltx::Limits, + scratch: &Path, +) -> Result> { + build_recovery_overlays_file_backed_stream(frames.into_iter().map(Ok), bases, limits, scratch) +} + +/// Streaming variant that retains no complete follower witness in memory. +pub fn build_recovery_overlays_file_backed_stream( + frames: I, + bases: &[RecoveryBase], + limits: cellule_ltx::Limits, + scratch: &Path, +) -> Result> +where + I: IntoIterator>, +{ + let mut frames = frames.into_iter(); + let first = frames + .next() + .ok_or(Error::Node("recovery witness is empty"))??; + let scope = first.scope(); + let mut builder = StreamingRecovery::new( + bases, + scope.leader_session, + scope.log_epoch, + limits, + scratch, + None, + )?; + for frame in std::iter::once(Ok(first)).chain(frames) { + builder.push(frame?)?; + } + builder.finish() +} diff --git a/crates/cellule-runtime/src/node/log_recovery/mod.rs b/crates/cellule-runtime/src/node/log_recovery/mod.rs index ebf4bbf2..a70493a5 100644 --- a/crates/cellule-runtime/src/node/log_recovery/mod.rs +++ b/crates/cellule-runtime/src/node/log_recovery/mod.rs @@ -13,9 +13,7 @@ use crate::control::authority::{CellAuthority, VersionedControl}; use crate::follower::FollowerReceipt; use crate::identity::NodeId; use crate::identity::{ApplicationId, CellId, Digest, SessionId}; -use crate::node::log::{ - RecoveryBase, build_recovery_overlays_file_backed, build_recovery_overlays_file_backed_stream, -}; +use crate::node::log::{RecoveryBase, StreamingRecovery}; use crate::node::log_state::NodeLogPhase; use crate::node::log_transport::{NodeLogTransport, SealRequest, TailRequest}; use crate::node::{FencedNodeSession, NodeDirectory, NodeTakeoverProof, SealedNodeLog}; @@ -230,6 +228,7 @@ pub struct NodeLogRecovery { limits: cellule_ltx::Limits, recovery_disk: cellule_ltx::DiskBudget, recovery_scratch: Option, + bundle_head: Option, } /// One dead-session Cell control that may need a recovered tail attached. @@ -372,6 +371,34 @@ pub async fn recoverable_cells_from_scopes_with_summary( scopes.insert(scope.cell, key); } } + // Follower scopes omit Cells whose whole suffix is selected in origin. + // Complete authenticated binding discovery is mandatory even for an empty + // follower witness; it never grants takeover authority. + for (binding_application, binding) in + crate::node::bundle::recovery::inventory_for_owner(authority.layout(), owner).await? + { + if binding_application.as_bytes() != &application { + continue; + } + let key = (*binding.incarnation.as_bytes(), binding.epoch); + let cell = *binding.cell.as_bytes(); + match scopes.get(&cell) { + Some(existing) if *existing != key => { + return Err(Error::Control( + "recovery Cell scope has multiple generations", + )); + } + Some(_) => {} + None => { + if scopes.len() == limit { + return Err(Error::Node( + "node recovery Cell inventory exceeds its limit", + )); + } + scopes.insert(cell, key); + } + } + } if scopes.is_empty() { return Ok(RecoverableCellInventory { cells: Vec::new(), @@ -512,29 +539,85 @@ impl RecoveryCoordinator { }); } - if sealed.frame_count() == 0 { - return Ok(RecoveryCoordinatorResult { - controls: Vec::new(), - publication: crate::recovery::manifest::RecoveryPublicationSummary::default(), - }); + if sealed.leader_session != self.recovery.leader_session + || sealed.log_epoch != self.recovery.log_epoch + || sealed.tiered_through != self.recovery.tiered_through + || sealed.durable_through.checked_sub(sealed.tiered_through) + != Some(sealed.frame_count()) + { + return Err(Error::Fenced); + } + let inventory = + crate::node::bundle::recovery::fenced_inventory(self.manifests.layout(), &fenced) + .await?; + for (application, binding) in &inventory { + let cell = cells + .iter() + .find(|cell| { + cell.application == *application && cell.observed.value().cell == binding.cell + }) + .ok_or(Error::Node( + "selected bundle Cell is absent from recovery inventory", + ))?; + let current = cell.observed.value(); + if current.bundle_binding != binding.bundle_binding + || current.epoch != binding.epoch + || current.incarnation != binding.incarnation + || current.code != binding.code + || current.schema != binding.schema + { + return Err(Error::Fenced); + } } let scratch = self.manifests.recovery_scratch_directory(); - let tails = if let Some(witness) = &sealed.witness { - let reader = witness.reader(self.recovery.limits)?; - build_recovery_overlays_file_backed_stream( - reader, - &bases, - self.recovery.limits, - &scratch, - )? + let mut builder = StreamingRecovery::new( + &bases, + *fenced.session().as_bytes(), + self.recovery.log_epoch, + self.recovery.limits, + &scratch, + Some(self.recovery.recovery_disk.clone()), + )?; + for cell in &cells { + if cell.observed.value().bundle_binding.is_some() { + builder.seed( + crate::node::bundle::recovery::selected_frames( + self.manifests.layout(), + &cell.authority, + &cell.observed, + &fenced, + self.recovery.limits, + ) + .await?, + )?; + } + } + let required_first = sealed + .tiered_through + .checked_add(1) + .ok_or(Error::Capacity("recovery witness range"))?; + let mut first_sequence = None; + let mut last_sequence = None; + if let Some(witness) = &sealed.witness { + for frame in witness.reader(self.recovery.limits)? { + let frame = frame?; + first_sequence.get_or_insert(frame.scope().node_sequence); + last_sequence = Some(frame.scope().node_sequence); + builder.push(frame)?; + } } else { - build_recovery_overlays_file_backed( - sealed.frames, - &bases, - self.recovery.limits, - &scratch, - )? - }; + for frame in sealed.frames { + first_sequence.get_or_insert(frame.scope().node_sequence); + last_sequence = Some(frame.scope().node_sequence); + builder.push(frame)?; + } + } + if first_sequence.is_some_and(|first| first != required_first) + || last_sequence.is_some_and(|last| last != sealed.durable_through) + { + return Err(Error::Node("recovery follower witness range differs")); + } + let tails = builder.finish()?; if tails.is_empty() { return Ok(RecoveryCoordinatorResult { controls: Vec::new(), @@ -690,6 +773,7 @@ impl NodeLogRecovery { limits, recovery_disk: default_recovery_disk(limits), recovery_scratch: None, + bundle_head: None, }) } @@ -734,6 +818,7 @@ impl NodeLogRecovery { ) .map(|mut recovery| { recovery.recovery_disk = recovery_disk; + recovery.bundle_head = fenced.bundle_head(); recovery }) } @@ -759,6 +844,7 @@ impl NodeLogRecovery { || log.members() != self.members || log.tiered_through() != self.tiered_through || log.active() != self.active + || fenced.bundle_head() != self.bundle_head || claim.claimant() != fenced.claimant() || claim.generation() != fenced.claim_generation() || claim.expires_at_ms() != fenced.claim_expires_at_ms() diff --git a/crates/cellule-runtime/src/recovery/manifest/mod.rs b/crates/cellule-runtime/src/recovery/manifest/mod.rs index 9cced140..50c8599c 100644 --- a/crates/cellule-runtime/src/recovery/manifest/mod.rs +++ b/crates/cellule-runtime/src/recovery/manifest/mod.rs @@ -13,6 +13,7 @@ use crate::node::log::RecoveredCellTail; use crate::{Error, Result}; const MAX_MANIFEST_BYTES: u64 = 2 << 20; +const MAX_MANIFEST_CELLS: usize = 4_096; const MULTIPART_BYTES: usize = 8 << 20; mod inventory; @@ -190,6 +191,9 @@ pub struct RecoveryManifestStore { } impl RecoveryManifestStore { + pub(crate) fn layout(&self) -> &CellStorageLayout { + &self.layout + } /// Creates a manifest store over one layout, with recovery disk bounded by /// the plan byte limit. #[must_use] @@ -265,6 +269,7 @@ impl RecoveryManifestStore { if leader_session.as_bytes().iter().all(|byte| *byte == 0) || log_epoch == 0 || tails.is_empty() + || tails.len() > MAX_MANIFEST_CELLS { return Err(Error::Node("invalid recovery manifest scope")); } @@ -278,7 +283,9 @@ impl RecoveryManifestStore { let runtime_predecessor = RootRef::from_ltx(cell, incarnation, predecessor)?; let final_position = tail.overlay.final_position(); let final_commit_sequence = tail.overlay.final_commit_sequence(); - let bundle = tail.overlay.into_bundle()?; + // Keep scratch admission through upload and optional cache work. + // Borrowing also permits file-backed admitted recovery overlays. + let bundle = tail.overlay.bundle(); summary.bundle_bytes = summary .bundle_bytes .checked_add(bundle.len()) @@ -289,7 +296,7 @@ impl RecoveryManifestStore { log_epoch, &bundle_digest, ); - publish_bundle_immutable(&self.layout, &path, &bundle, &mut summary).await?; + publish_bundle_immutable(&self.layout, &path, bundle, &mut summary).await?; if self.artifact_store.is_some() { let key = RecoveryArtifactKey::new( leader_session, @@ -305,7 +312,7 @@ impl RecoveryManifestStore { final_commit_sequence, Digest::from_bytes(bundle_digest), ); - artifacts.push((key, bundle)); + artifacts.push((key, tail.overlay)); } rows.push(ManifestCell { application: tail.application, @@ -358,11 +365,17 @@ impl RecoveryManifestStore { ); publish_immutable(&self.layout, &path, &body, MAX_MANIFEST_BYTES, &mut summary).await?; if let Some(store) = &self.artifact_store { - for (key, bundle) in artifacts { + for (key, overlay) in artifacts { let store = Arc::clone(store); // Both immutable objects are the correctness boundary; a local // cache admission failure must not block control progress. - let _ = tokio::task::spawn_blocking(move || store.retain(key, bundle)).await; + let (bundle, owner) = overlay.into_bundle_with_owner(); + let _ = tokio::task::spawn_blocking(move || { + let result = store.retain(key, bundle); + drop(owner); + result + }) + .await; } } Ok(PinnedRecoveryCells { @@ -636,7 +649,7 @@ impl TryFrom for RecoveryManifest { type Error = Error; fn try_from(raw: RawManifest) -> Result { - if raw.version != 1 || raw.cells.is_empty() || raw.cells.len() > 1_024 { + if raw.version != 1 || raw.cells.is_empty() || raw.cells.len() > MAX_MANIFEST_CELLS { return Err(Error::Node("invalid recovery manifest shape")); } let leader_session = SessionId::from_bytes(unhex(&raw.leader_session)?); diff --git a/crates/cellule-runtime/src/recovery/manifest/tests.rs b/crates/cellule-runtime/src/recovery/manifest/tests.rs index 3a8cbe15..73cde902 100644 --- a/crates/cellule-runtime/src/recovery/manifest/tests.rs +++ b/crates/cellule-runtime/src/recovery/manifest/tests.rs @@ -26,6 +26,141 @@ async fn recovery_fixture() -> RecoveryFixture { recovery_fixture_with_store(None).await } +struct PausedArtifactStore { + started: tokio::sync::Notify, + release: Mutex>, +} + +impl RecoveryArtifactStore for PausedArtifactStore { + fn retain(&self, _key: RecoveryArtifactKey, bundle: cellule_ltx::bundle::Bundle) -> Result<()> { + // Ownership stays unique so the ordinary file cache can take it without + // copying or dropping source admission before its own work has joined. + let path = bundle.detach_file()?; + self.started.notify_one(); + self.release + .lock() + .unwrap() + .recv_timeout(std::time::Duration::from_secs(10)) + .unwrap(); + std::fs::remove_file(path)?; + Ok(()) + } + fn load(&self, _key: &RecoveryArtifactKey) -> Result> { + Ok(None) + } +} + +#[tokio::test] +async fn cancelled_pin_retains_original_scratch_admission_through_cache_job() { + let fixture = recovery_fixture().await; + let disk = cellule_ltx::DiskBudget::new(32 << 20); + let loader = fixture.manifests.clone().with_recovery_disk(disk.clone()); + let pin = fixture.pinned; + let overlay = loader + .load_overlay(pin.cell, pin.incarnation, &pin.recovery) + .await + .unwrap(); + let bytes = overlay.bundle().len(); + assert_eq!(disk.used(), bytes); + let (release, receiver) = std::sync::mpsc::channel(); + let cache = Arc::new(PausedArtifactStore { + started: tokio::sync::Notify::new(), + release: Mutex::new(receiver), + }); + let manifests = fixture.manifests.with_recovery_artifacts(cache.clone()); + let leader = pin.recovery.leader_session; + let epoch = pin.recovery.log_epoch; + let task = tokio::spawn(async move { + manifests + .pin( + leader, + epoch, + vec![crate::node::log::RecoveredCellTail { + application: *pin.application.as_bytes(), + cell_epoch: pin.cell_epoch, + first_node_sequence: pin.recovery.first_node_sequence, + last_node_sequence: pin.recovery.last_node_sequence, + overlay, + }], + ) + .await + }); + tokio::time::timeout(std::time::Duration::from_secs(10), cache.started.notified()) + .await + .unwrap(); + task.abort(); + assert!(matches!(task.await, Err(error) if error.is_cancelled())); + assert_eq!( + disk.used(), + bytes, + "cancelled waiter cannot release accepted cache work" + ); + release.send(()).unwrap(); + tokio::time::timeout(std::time::Duration::from_secs(10), async { + while disk.used() != 0 { + tokio::time::sleep(std::time::Duration::from_millis(1)).await; + } + }) + .await + .unwrap(); +} + +#[tokio::test] +async fn manifest_codec_covers_2000_cells_with_the_existing_byte_bound() { + let fixture = recovery_fixture().await; + let recovery = fixture.pinned.recovery; + let path = fixture.layout.node_log_recovery_path( + recovery.leader_session.as_bytes(), + recovery.log_epoch, + recovery.manifest_digest.as_bytes(), + ); + let (body, _) = fixture + .layout + .store() + .get_with_etag_bounded(&path, MAX_MANIFEST_BYTES) + .await + .unwrap(); + let original = RecoveryManifest::decode(&body).unwrap(); + let row = &original.cells[0]; + let cells = (1_u64..=2_000) + .map(|i| { + let mut cell = [0_u8; 32]; + cell[24..].copy_from_slice(&i.to_be_bytes()); + let mut predecessor = row.predecessor; + predecessor.cell = cell; + ManifestCell { + application: row.application, + cell, + incarnation: row.incarnation, + cell_epoch: row.cell_epoch, + first_node_sequence: i, + last_node_sequence: i, + predecessor, + final_position: row.final_position, + final_commit_sequence: row.final_commit_sequence, + bundle_digest: row.bundle_digest, + } + }) + .collect(); + let manifest = RecoveryManifest { + leader_session: original.leader_session, + log_epoch: original.log_epoch, + cells, + }; + let body = manifest.encode().unwrap(); + assert!((body.len() as u64) < MAX_MANIFEST_BYTES); + assert_eq!(RecoveryManifest::decode(&body).unwrap().cells.len(), 2_000); + let mut raw = RawManifest::from(&manifest); + let duplicate = serde_json::to_vec(&raw.cells[0]).unwrap(); + raw.cells.resize_with(MAX_MANIFEST_CELLS + 1, || { + serde_json::from_slice(&duplicate).unwrap() + }); + assert!(matches!( + RecoveryManifest::try_from(raw), + Err(Error::Node("invalid recovery manifest shape")) + )); +} + struct MemoryArtifactStore { limits: cellule_ltx::Limits, reject_retain: bool, diff --git a/docs/bundle-coverage-implementation.md b/docs/bundle-coverage-implementation.md index 638c0a81..79a0aba2 100644 --- a/docs/bundle-coverage-implementation.md +++ b/docs/bundle-coverage-implementation.md @@ -42,6 +42,16 @@ Uploading an immutable object supplies no coverage proof. | Node withdrawal/maintenance | Refuse unresolved bindings; stale fencing preserves the catalog head; one bundle-bound boot cannot rotate its native log to another epoch | | Backup and collection | Backup refuses bound Cells. Coverage objects have no deletion path; this is retention, not a qualified collection implementation | +Failed-owner recovery now joins dependency-verified selected prefixes and the +sealed follower witness in the same file-backed builder. Complete shard +inventory includes selected-only Cells; identical native overlap is skipped, +while conflicting bytes and an omitted Cell reject before attachment. A bound +control may retain only its original recovery session/epoch after canonical +fencing. Its pin remains until complete terminal materialization/checkpoint; +the lower-level node seal enforces that obligation too. Failed-boot materializer +scheduling, terminal catalog CAS and provisional enrollment resolution remain +unfinished. This does not enable ordinary bundle responses. + The original SQL/capture/submission jobs must join before `close_cell_issuance`. Its ordered gate prevents late assignment from consuming a node sequence. The legacy identity-free `DurabilityGate::issue` cannot produce a per-Cell closure: @@ -161,6 +171,16 @@ These results verify correctness and component work; the latest source has no new end-to-end TPS measurement. Ordinary application responses still use per-Cell publication, and production bundle ACK integration remains unfinished. +The combined selected-prefix/follower recovery snapshot passed all twelve +contributor checks: **1,916 workspace tests passed, 38 ignored; 60 local LTX +tests passed**. Its five new two-Cell cases use real managed captures and +file-backed follower fsync, verifying a pruned prefix with a prior Fleet ACK, +identical overlap, selected-only recovery, conflicting overlap and omitted +inventory. Separate tests cover a 2,000-scope manifest codec and retention of +scratch admission when a cancelled waiter leaves cache work running. The scope +codec is a metadata fixture, not 2,000 writers or throughput evidence. Failed-boot +root materialization and terminal catalog closure remain required. + The following counts describe the earlier inline-index snapshot, not the latest application TPS: @@ -247,11 +267,10 @@ Remaining work before responses can use bundle proof: 2. Build on the authenticated index: bound admitted maintenance inventory, increase checkpoint density with retained-byte accounting, and measure the complete materialization/checkpoint/collection cost. -3. Fence and seal a failed original node's complete follower-issued suffix, - including ACKs above the selected head, before reconstructing and closing - every original binding, including interrupted provisional enrollments. - Current departure/maintenance guards refuse this - unfinished recovery rather than advancing a replacement writer. +3. Complete failed-boot materialization and terminal catalog selection after + combined prefix/follower reconstruction, including prior Fleet ACKs and + interrupted provisional enrollments. Current departure and canonical seal + guards refuse unfinished closure rather than advancing a replacement writer. 4. Implement complete cross-Cell reference inventory, pins and grace-qualified collection. Preserve original catalog and base dependencies throughout. 5. Run the unchanged all-ACK cold recovery, paired Docker throughput/latency, From 13ab0ec19925f4ad9ba5a1b0d0ecc8bda85a406e Mon Sep 17 00:00:00 2001 From: forhappy Date: Wed, 7 Oct 2026 19:40:43 -0700 Subject: [PATCH 022/102] Materialize fenced bundle recovery and atomically close its catalog --- .../docs/failover-and-followers.md | 17 +- .../docs/write-performance-design.md | 11 +- .../src/control/authority/mod.rs | 11 +- crates/cellule-runtime/src/control/mod.rs | 6 +- .../cellule-runtime/src/node/bundle/proof.rs | 15 +- .../src/node/bundle/recovery/drain.rs | 372 ++++++++++++++++++ .../bundle/{recovery.rs => recovery/mod.rs} | 70 +++- .../src/node/bundle/tests/faults.rs | 26 +- .../src/node/bundle/tests/index/density.rs | 13 +- .../src/node/bundle/tests/lifecycle.rs | 52 +++ .../src/node/bundle/tests/mod.rs | 22 ++ .../src/node/bundle/tests/ranges.rs | 12 +- .../src/node/bundle/tests/recovery/cohort.rs | 185 +++++++++ .../tests/{recovery.rs => recovery/mod.rs} | 317 ++++++++++++++- .../src/node/log_recovery/mod.rs | 179 ++++++--- .../src/publication/lineage.rs | 6 +- crates/cellule-runtime/src/publication/mod.rs | 2 +- .../src/recovery/manifest/mod.rs | 15 +- .../src/recovery/manifest/reconcile.rs | 67 ++++ docs/bundle-coverage-implementation.md | 43 +- 20 files changed, 1325 insertions(+), 116 deletions(-) create mode 100644 crates/cellule-runtime/src/node/bundle/recovery/drain.rs rename crates/cellule-runtime/src/node/bundle/{recovery.rs => recovery/mod.rs} (68%) create mode 100644 crates/cellule-runtime/src/node/bundle/tests/recovery/cohort.rs rename crates/cellule-runtime/src/node/bundle/tests/{recovery.rs => recovery/mod.rs} (51%) create mode 100644 crates/cellule-runtime/src/recovery/manifest/reconcile.rs diff --git a/crates/cellule-runtime/docs/failover-and-followers.md b/crates/cellule-runtime/docs/failover-and-followers.md index 2874a759..b5077160 100644 --- a/crates/cellule-runtime/docs/failover-and-followers.md +++ b/crates/cellule-runtime/docs/failover-and-followers.md @@ -1577,9 +1577,20 @@ before writing, and cache jobs retain that admission after caller cancellation. A bound Cell can pin only recovery from its original session and log epoch, after canonical node fencing. Its binding remains pinned. The canonical node seal refuses until every binding has a terminal materialized checkpoint; -uploading or attaching a manifest grants no transfer authority. Scheduling that -failed-boot materialization, interrupted enrollment resolution and terminal -catalog CAS is still unfinished, so ordinary bundle ACKs remain disabled. The +uploading or attaching a manifest grants no transfer authority. The public +`recover_and_seal` path materializes every original bound root through ordinary +lineage and Cell CAS, including quiet bindings. It stages bounded catalog cohorts +and selects their complete terminal inventory with one recovery-claim node CAS +before log seal. Claims are rechecked after I/O; exact lost replies reconcile. +Partial-root retries reuse the original digest-verified manifest and match every +remaining tail's scope, predecessor, complete physical/logical range and bundle +digest. A materialized root ahead of the old catalog is only a verified base; +the full witness and terminal endpoint must still agree. + +Closed original roots also supply verified recovery bases after transfer, so +interruption between terminal catalog CAS and log seal does not reload a +successor as the failed writer. Provisional enrollment resolution and admitted +host scheduling remain unfinished; ordinary bundle ACKs remain disabled. The manifest reader admits up to 4,096 scopes under the existing 2 MiB byte ceiling; the 2,000-scope codec test is format evidence, not node performance qualification. diff --git a/crates/cellule-runtime/docs/write-performance-design.md b/crates/cellule-runtime/docs/write-performance-design.md index e69db42a..673dd87b 100644 --- a/crates/cellule-runtime/docs/write-performance-design.md +++ b/crates/cellule-runtime/docs/write-performance-design.md @@ -107,6 +107,11 @@ The recovery coordinator now discovers selected-only Cells and joins exact origin prefixes with the complete sealed follower witness in one admitted file-backed reconstruction path. It rejects conflicting overlap and incomplete Cell inventory. Bound overlays retain the original pin, and the canonical node -seal refuses unfinished materialization/checkpoint closure. Failed-boot root -materialization, terminal catalog selection and provisional enrollment resolution -remain required before enabling the canonical coverage frontier or actor ACKs. +seal refuses unfinished materialization/checkpoint closure. `recover_and_seal` +now materializes original bound roots through ordinary lineage and exact Cell +CAS, then stages bounded catalog cohorts and selects their complete terminal +inventory with one recovery-claim node CAS. Quiet Cells participate. Exact +manifest reuse supports partial-root retries, and closed original roots support +resumption after transfer or interrupted log seal. Provisional enrollment +resolution, admitted host scheduling and full qualification remain required +before enabling the canonical coverage frontier or actor ACKs. diff --git a/crates/cellule-runtime/src/control/authority/mod.rs b/crates/cellule-runtime/src/control/authority/mod.rs index d8764f3b..0ed50c57 100644 --- a/crates/cellule-runtime/src/control/authority/mod.rs +++ b/crates/cellule-runtime/src/control/authority/mod.rs @@ -180,11 +180,18 @@ impl CellAuthority { if transition == Transition::BindBundle { crate::node::bundle::store::ensure_enrollment(&self.layout, &next).await?; } - if transition == Transition::AttachRecovery && observed.value.bundle_binding.is_some() { + if matches!( + transition, + Transition::AttachRecovery | Transition::PublishRecovery + ) && observed.value.bundle_binding.is_some() + { crate::node::bundle::recovery::ensure_attachment( &self.layout, &observed.value, - next.recovery.as_ref().ok_or(Error::Fenced)?, + next.recovery + .as_ref() + .or(observed.value.recovery.as_ref()) + .ok_or(Error::Fenced)?, ) .await?; } diff --git a/crates/cellule-runtime/src/control/mod.rs b/crates/cellule-runtime/src/control/mod.rs index 91dddcca..fc0c68ce 100644 --- a/crates/cellule-runtime/src/control/mod.rs +++ b/crates/cellule-runtime/src/control/mod.rs @@ -635,8 +635,10 @@ impl Control { let Some(recovery) = self.recovery.as_ref() else { return Err(Error::Control("invalid recovery publication")); }; - if self.state != ControlState::Recovering - || next.state != ControlState::Recovering + let original_bound_writer = + self.state == ControlState::Serving && self.bundle_binding.is_some(); + if (self.state != ControlState::Recovering && !original_bound_writer) + || next.state != self.state || next.recovery.is_some() || self.epoch != next.epoch || self.owner != next.owner diff --git a/crates/cellule-runtime/src/node/bundle/proof.rs b/crates/cellule-runtime/src/node/bundle/proof.rs index 3a900a12..2c0f395c 100644 --- a/crates/cellule-runtime/src/node/bundle/proof.rs +++ b/crates/cellule-runtime/src/node/bundle/proof.rs @@ -53,6 +53,19 @@ pub(super) async fn load_coverage_at( authority: &CellAuthority, control: &VersionedControl, limits: cellule_ltx::Limits, +) -> Result<(BundleCoverageProof, Vec)> { + let (proof, frames) = load_binding_at(layout, head, authority, control, limits).await?; + let current = control.value().ltx_root().ok_or(Error::Fenced)?; + checkpoint_prefix(&proof.binding, &frames, ¤t)?; + Ok((proof, frames)) +} + +pub(super) async fn load_binding_at( + layout: &cellule_ltx::CellStorageLayout, + head: NodeBundleHead, + authority: &CellAuthority, + control: &VersionedControl, + limits: cellule_ltx::Limits, ) -> Result<(BundleCoverageProof, Vec)> { let value = control.value(); let pin = value.bundle_binding.ok_or(Error::PendingPublication)?; @@ -82,8 +95,6 @@ pub(super) async fn load_coverage_at( return Err(Error::Fenced); } let frames = verify_binding(layout, pin.session, head.epoch, &binding, limits).await?; - let current = value.ltx_root().ok_or(Error::Fenced)?; - checkpoint_prefix(&binding, &frames, ¤t)?; Ok(( BundleCoverageProof { pin, diff --git a/crates/cellule-runtime/src/node/bundle/recovery/drain.rs b/crates/cellule-runtime/src/node/bundle/recovery/drain.rs new file mode 100644 index 00000000..e9aace8a --- /dev/null +++ b/crates/cellule-runtime/src/node/bundle/recovery/drain.rs @@ -0,0 +1,372 @@ +//! Terminal catalog selection owned by a verified complete recovery attempt. +use super::*; +use crate::node::log_recovery::RecoveryCell; +use crate::recovery::manifest::RecoveryManifestStore; +use std::collections::BTreeMap; + +type CellKey = ([u8; 16], [u8; 32]); +type Endpoint = (u64, u64, cellule_ltx::Position); + +struct Scope { + binding: Binding, + terminal: Endpoint, + cell: RecoveryCell, +} + +/// Never constructed from caller-supplied watermarks. The coordinator owns it +/// while verifying every selected dependency and every sealed follower frame. +pub(crate) struct BundleRecoveryDrain { + original: NodeBundleHead, + scopes: BTreeMap, + closed: BTreeMap, + durable_through: u64, +} + +impl BundleRecoveryDrain { + pub(crate) async fn open( + layout: &cellule_ltx::CellStorageLayout, + fenced: &FencedNodeSession, + cells: &[RecoveryCell], + durable_through: u64, + limits: cellule_ltx::Limits, + ) -> Result> { + let Some(original) = fenced.bundle_head() else { + return Ok(None); + }; + if durable_through < original.selected_through { + return Err(Error::Node( + "sealed recovery does not cover selected bundle", + )); + } + let mut scopes = BTreeMap::new(); + let mut closed = BTreeMap::new(); + for binding in fenced_bindings(layout, fenced).await? { + if binding.phase == BindingPhase::Closed { + if binding.terminal + != Some(( + binding.selected_sequence, + binding.selected_commit, + binding.selected_position, + )) + || binding.control.recovery.is_some() + || !binding.locators.is_empty() + { + return Err(Error::PendingPublication); + } + verify_base(layout, &binding, limits).await?; + closed.insert( + ( + *binding.application.as_bytes(), + *binding.control.cell.as_bytes(), + *binding.control.incarnation.as_bytes(), + binding.control.epoch, + ), + binding, + ); + continue; + } + // Interrupted enrollment still needs explicit reconciliation. It + // must never disappear merely because no follower row names it. + if binding.phase == BindingPhase::Provisional { + return Err(Error::PendingPublication); + } + let cell = cells + .iter() + .find(|cell| { + cell.application == binding.application + && cell.observed.value().cell == binding.control.cell + }) + .ok_or(Error::Node( + "selected bundle Cell is absent from recovery inventory", + ))?; + let current = cell.observed.value(); + if !same_scope(&binding.control, current) || current.owner != binding.control.owner { + return Err(Error::Fenced); + } + let terminal = ( + binding.selected_sequence, + binding.selected_commit, + binding.selected_position, + ); + if scopes + .insert( + ( + *binding.application.as_bytes(), + *binding.control.cell.as_bytes(), + ), + Scope { + binding, + terminal, + cell: cell.clone(), + }, + ) + .is_some() + { + return Err(Error::Node("bundle inventory repeats an active Cell")); + } + } + Ok(Some(Self { + original, + scopes, + closed, + durable_through, + })) + } + + pub(crate) fn observe( + &mut self, + native: cellule_ltx::NodeFrameScope, + position: cellule_ltx::Position, + ) -> Result<()> { + if let Some(closed) = self.closed.get(&( + native.application, + native.cell, + native.incarnation, + native.cell_epoch, + )) { + if native.node_sequence > closed.selected_sequence + || native.commit_sequence > closed.selected_commit + || position.txid > closed.selected_position.txid + { + return Err(Error::Node("sealed frame exceeds closed original binding")); + } + return Ok(()); + } + let scope = self + .scopes + .get_mut(&(native.application, native.cell)) + .ok_or(Error::Node("sealed bundle frame has no original binding"))?; + if native.incarnation != *scope.binding.control.incarnation.as_bytes() + || native.cell_epoch != scope.binding.control.epoch + { + return Err(Error::Fenced); + } + if native.node_sequence > scope.terminal.0 { + scope.terminal = (native.node_sequence, native.commit_sequence, position); + } + Ok(()) + } + + pub(crate) fn add_closed_bases( + &self, + bases: &mut Vec, + ) -> Result<()> { + for (key, binding) in &self.closed { + if bases.iter().any(|base| { + ( + base.application, + base.root.cell, + base.root.incarnation, + base.cell_epoch, + ) == *key + }) { + continue; + } + bases.push(crate::node::log::RecoveryBase { + application: key.0, + cell_epoch: key.3, + root: binding + .control + .ltx_root() + .ok_or(Error::PendingPublication)?, + }); + } + Ok(()) + } + + /// Materialize every original bound Cell before changing any catalog row. + /// A late failure retains original pins and immutable recovery pointers. + pub(crate) async fn complete( + self, + directory: &crate::node::NodeDirectory, + manifests: &RecoveryManifestStore, + fenced: &FencedNodeSession, + now_ms: i64, + started: std::time::Instant, + ) -> Result<(FencedNodeSession, Vec)> { + if fenced.bundle_head() != Some(self.original) + || directory.layout.node_path(fenced.session().as_bytes()) + != manifests.layout().node_path(fenced.session().as_bytes()) + || directory.layout.immutable_cache_identity() + != manifests.layout().immutable_cache_identity() + { + return Err(Error::Fenced); + } + let mut controls = Vec::with_capacity(self.scopes.len()); + for scope in self.scopes.values() { + check_claim(directory, fenced, logical_now(now_ms, started)?).await?; + let authority = &scope.cell.authority; + let mut current = authority + .load(scope.binding.control.cell) + .await? + .ok_or(Error::Fenced)?; + if !same_scope(&scope.binding.control, current.value()) + || current.value().owner != scope.binding.control.owner + { + return Err(Error::Fenced); + } + if let Some(recovery) = ¤t.value().recovery { + let store = manifests.for_application(scope.binding.application); + let overlay = store + .load_overlay(current.value().cell, current.value().incarnation, recovery) + .await?; + let replica = cellule_ltx::CellReplica::new( + authority.layout().clone(), + *current.value().cell.as_bytes(), + *current.value().incarnation.as_bytes(), + manifests.limits(), + )?; + let (replica, _) = crate::publication::lineage::replica(replica, authority); + let prepared = replica + .prepare_recovered_overlay(&overlay, current.value().schema) + .await + .map_err(crate::publication::lineage::error)?; + let successor = current + .value() + .publish_recovery(&prepared, current.value().next_due_ms)?; + check_claim(directory, fenced, logical_now(now_ms, started)?).await?; + current = match authority + .transition( + ¤t, + successor.clone(), + crate::control::Transition::PublishRecovery, + ) + .await + { + Ok(published) => published, + Err(source) => { + let actual = authority + .load(scope.binding.control.cell) + .await? + .ok_or(Error::Fenced)?; + if actual.value() != &successor { + return Err(source); + } + actual + } + }; + } + let root = current + .value() + .ltx_root() + .ok_or(Error::PendingPublication)?; + if root.commit_sequence != scope.terminal.1 + || root.position != scope.terminal.2 + || current.value().recovery.is_some() + { + return Err(Error::PendingPublication); + } + let mut materialized = scope.binding.clone(); + materialized.control = current.value().clone(); + verify_base(manifests.layout(), &materialized, manifests.limits()).await?; + controls.push(current); + } + let entries = self + .scopes + .into_values() + .zip(controls.iter()) + .collect::>(); + let mut staged = self.original; + for cohort in entries.chunks(MAX_FRAMES) { + let head = staged; + let keys = cohort + .iter() + .map(|(scope, _)| { + ( + *scope.binding.application.as_bytes(), + *scope.binding.control.cell.as_bytes(), + ) + }) + .collect(); + let mut catalog = + store::load_catalog_cells(manifests.layout(), fenced.session(), head, &keys) + .await?; + for (scope, control) in cohort { + let pin = scope.binding.control.bundle_binding.ok_or(Error::Fenced)?; + let binding = catalog.binding_mut(pin.digest)?; + if !same_scope(&binding.control, control.value()) + || binding.phase == BindingPhase::Provisional + || binding.selected_sequence > scope.terminal.0 + { + return Err(Error::Fenced); + } + binding.control = control.value().clone(); + binding.phase = BindingPhase::Closed; + binding.terminal = Some(scope.terminal); + binding.selected_sequence = scope.terminal.0; + binding.selected_commit = scope.terminal.1; + binding.selected_position = scope.terminal.2; + binding.first_commit = scope.terminal.1; + binding.locators.clear(); + } + // Every original root was verified above, including quiet Cells. + // Only this complete witness may extend a failed boot's frontier. + catalog.selected_through = self.durable_through; + let prepared = directory.upload_catalog(Some(head), catalog, &[]).await?; + staged = prepared.head; + check_claim(directory, fenced, logical_now(now_ms, started)?).await?; + } + // Intermediate cohort objects are immutable staging, never authority. + // Select the fully closed inventory with one original-claim node CAS. + store::ensure_session_drained(manifests.layout(), fenced.session(), Some(staged)).await?; + let (mut current, token) = + check_claim(directory, fenced, logical_now(now_ms, started)?).await?; + if staged == self.original { + return Ok((fenced.clone(), controls)); + } + current.bundle = Some(staged); + let path = manifests.layout().node_path(fenced.session().as_bytes()); + let fresh = current.fenced()?; + if let Err(source) = manifests + .layout() + .store() + .update(&path, Bytes::from(current.encode()?), token) + .await + { + // Lost reply only reconciles this exact immutable head, original + // claim generation, expiry and recovering log. + check_claim(directory, &fresh, logical_now(now_ms, started)?) + .await + .map_err(|_| Error::from(source))?; + } + Ok((fresh, controls)) + } +} + +fn same_scope(original: &Control, current: &Control) -> bool { + original.cell == current.cell + && original.incarnation == current.incarnation + && original.epoch == current.epoch + && original.bundle_binding == current.bundle_binding + && original.code == current.code + && original.schema == current.schema +} + +async fn check_claim( + directory: &crate::node::NodeDirectory, + fenced: &FencedNodeSession, + now_ms: i64, +) -> Result<(crate::node::directory::NodeTombstone, cellule_store::ETag)> { + directory + .load(fenced.claimant(), now_ms) + .await? + .ok_or(Error::Fenced)?; + let path = directory.layout.node_path(fenced.session().as_bytes()); + let Some((NodeRecord::Tombstone(current), token)) = directory.load_record_at(&path).await? + else { + return Err(Error::Fenced); + }; + if current.fenced()? != *fenced || now_ms >= fenced.claim_expires_at_ms() { + return Err(Error::Fenced); + } + Ok((*current, token)) +} + +pub(crate) fn logical_now(now_ms: i64, started: std::time::Instant) -> Result { + now_ms + .checked_add( + i64::try_from(started.elapsed().as_millis()) + .map_err(|_| Error::Capacity("recovery clock duration"))?, + ) + .ok_or(Error::Capacity("recovery clock overflow")) +} diff --git a/crates/cellule-runtime/src/node/bundle/recovery.rs b/crates/cellule-runtime/src/node/bundle/recovery/mod.rs similarity index 68% rename from crates/cellule-runtime/src/node/bundle/recovery.rs rename to crates/cellule-runtime/src/node/bundle/recovery/mod.rs index 5f47da7c..61bef8c4 100644 --- a/crates/cellule-runtime/src/node/bundle/recovery.rs +++ b/crates/cellule-runtime/src/node/bundle/recovery/mod.rs @@ -1,8 +1,12 @@ //! Failed-session inventory and exact selected-prefix reconstruction. use super::*; +mod drain; use crate::control::authority::{CellAuthority, VersionedControl}; use crate::node::FencedNodeSession; use crate::node::directory::NodeRecord; +pub(crate) use drain::{BundleRecoveryDrain, logical_now}; + +pub(crate) type GenerationKey = ([u8; 16], [u8; 32], [u8; 16], u64); async fn record(layout: &cellule_ltx::CellStorageLayout, session: SessionId) -> Result { let (bytes, _) = layout @@ -25,15 +29,23 @@ async fn record(layout: &cellule_ltx::CellStorageLayout, session: SessionId) -> /// Discovery remains advisory; only a fenced claim can attach recovered state. /// Authenticate every shard without loading dense histories during discovery. +#[derive(Default)] +pub(crate) struct OwnerBundleInventory { + pub(crate) controls: Vec<(crate::ApplicationId, Control)>, + pub(crate) closed: std::collections::BTreeSet, +} + pub(crate) async fn inventory_for_owner( layout: &cellule_ltx::CellStorageLayout, session: SessionId, -) -> Result> { +) -> Result { let node = match record(layout, session).await { Ok(node) => node, // Scope-only catalog discovery also supports applications that do not // use a node directory. A bound Cell still fails the fenced lookup. - Err(Error::Storage(cellule_store::StorageError::NotFound { .. })) => return Ok(Vec::new()), + Err(Error::Storage(cellule_store::StorageError::NotFound { .. })) => { + return Ok(OwnerBundleInventory::default()); + } Err(source) => return Err(source), }; let head = match node { @@ -41,7 +53,7 @@ pub(crate) async fn inventory_for_owner( NodeRecord::Tombstone(node) => node.bundle, }; let Some(head) = head else { - return Ok(Vec::new()); + return Ok(OwnerBundleInventory::default()); }; inventory_at(layout, session, head).await } @@ -50,19 +62,29 @@ async fn inventory_at( layout: &cellule_ltx::CellStorageLayout, session: SessionId, head: NodeBundleHead, -) -> Result> { - Ok(index::binding_inventory(layout, session, head) - .await? - .into_iter() - .filter(|binding| binding.phase != BindingPhase::Closed) - .map(|binding| (binding.application, binding.control)) - .collect()) +) -> Result { + let mut inventory = OwnerBundleInventory::default(); + for binding in index::binding_inventory(layout, session, head).await? { + if binding.phase == BindingPhase::Closed { + inventory.closed.insert(( + *binding.application.as_bytes(), + *binding.control.cell.as_bytes(), + *binding.control.incarnation.as_bytes(), + binding.control.epoch, + )); + } else { + inventory + .controls + .push((binding.application, binding.control)); + } + } + Ok(inventory) } -pub(crate) async fn fenced_inventory( +async fn fenced_bindings( layout: &cellule_ltx::CellStorageLayout, fenced: &FencedNodeSession, -) -> Result> { +) -> Result> { let Some(head) = fenced.bundle_head() else { return Ok(Vec::new()); }; @@ -78,7 +100,7 @@ pub(crate) async fn fenced_inventory( { return Err(Error::Fenced); } - inventory_at(layout, fenced.session(), head).await + index::binding_inventory(layout, fenced.session(), head).await } pub(crate) async fn selected_frames( @@ -95,9 +117,25 @@ pub(crate) async fn selected_frames( } // Return original frames, including a materialized overlap, so the global // follower witness must agree byte-for-byte with selected native sequence. - super::proof::load_coverage_at(layout, head, authority, control, limits) - .await - .map(|(_, frames)| frames) + let (proof, frames) = + super::proof::load_binding_at(layout, head, authority, control, limits).await?; + let root = control.value().ltx_root().ok_or(Error::Fenced)?; + match super::proof::checkpoint_prefix(&proof.binding, &frames, &root) { + Ok(_) => {} + Err(Error::PendingPublication) + if root.commit_sequence > proof.commit_sequence() + && root.position.txid > proof.position().txid => {} + Err(source) => return Err(source), + } + // Recovery may resume between its root CAS and terminal catalog CAS. This + // ahead root is only a verified base; the full follower witness still has + // to agree and the complete terminal endpoint must match before closure. + if proof.binding.control.ltx_root() != Some(root) { + let mut current = proof.binding.clone(); + current.control = control.value().clone(); + super::proof::verify_base(layout, ¤t, limits).await?; + } + Ok(frames) } /// A bound writer may retain a recovery overlay only after canonical fencing. diff --git a/crates/cellule-runtime/src/node/bundle/tests/faults.rs b/crates/cellule-runtime/src/node/bundle/tests/faults.rs index a6d79874..a4c43356 100644 --- a/crates/cellule-runtime/src/node/bundle/tests/faults.rs +++ b/crates/cellule-runtime/src/node/bundle/tests/faults.rs @@ -4,12 +4,14 @@ use object_store::{ CopyOptions, GetOptions, GetResult, ListResult, MultipartUpload, ObjectMeta, ObjectStore, PutMultipartOptions, PutOptions, PutPayload, PutResult, }; -use std::sync::atomic::{AtomicU8, Ordering}; +use std::sync::atomic::{AtomicU8, AtomicUsize, Ordering}; #[derive(Debug, Default)] -struct ReplyFault { +pub(super) struct ReplyFault { inner: InMemory, - mode: AtomicU8, + pub(super) mode: AtomicU8, + pub(super) node_updates: AtomicUsize, + pub(super) coverage_puts: AtomicUsize, } impl std::fmt::Display for ReplyFault { fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { @@ -30,8 +32,23 @@ impl ObjectStore for ReplyFault { opts: PutOptions, ) -> object_store::Result { let mode = self.mode.load(Ordering::SeqCst); + if path.as_ref().ends_with("/control.json") + && matches!(opts.mode, object_store::PutMode::Update(_)) + { + if mode == 6 { + self.mode.store(7, Ordering::SeqCst); + } else if mode == 7 { + return Err(denied()); + } + } let node_update = path.as_ref().contains("/nodes/") && matches!(opts.mode, object_store::PutMode::Update(_)); + if node_update { + self.node_updates.fetch_add(1, Ordering::SeqCst); + } + if path.as_ref().ends_with(".cnb") { + self.coverage_puts.fetch_add(1, Ordering::SeqCst); + } if node_update && mode == 4 { self.mode .compare_exchange(4, 5, Ordering::SeqCst, Ordering::SeqCst) @@ -50,6 +67,9 @@ impl ObjectStore for ReplyFault { return Err(denied()); } let eligible = (mode == 1 && path.as_ref().ends_with(".cnb")) + || (mode == 8 + && path.as_ref().ends_with("/control.json") + && matches!(opts.mode, object_store::PutMode::Update(_))) || (mode == 2 && path.as_ref().contains("/nodes/") && matches!(opts.mode, object_store::PutMode::Update(_))); diff --git a/crates/cellule-runtime/src/node/bundle/tests/index/density.rs b/crates/cellule-runtime/src/node/bundle/tests/index/density.rs index 388941f2..637a145d 100644 --- a/crates/cellule-runtime/src/node/bundle/tests/index/density.rs +++ b/crates/cellule-runtime/src/node/bundle/tests/index/density.rs @@ -9,11 +9,12 @@ async fn dense_history_retains_215_exact_commands_before_checkpoint_and_cold_res let mut segments = Vec::new(); let mut metadata_bytes = 0; for commit in 2..=216 { + let now = f.heartbeat().await; let (cuts, frames, assignment) = f.append(&mut cell, commit); segments.extend(cuts.segments); let proposal = f .directory - .prepare_node_bundle(&f.node, &frames, &[assignment], NOW) + .prepare_node_bundle(&f.node, &frames, &[assignment], now) .await .unwrap(); metadata_bytes = proposal.body.len() @@ -23,7 +24,7 @@ async fn dense_history_retains_215_exact_commands_before_checkpoint_and_cold_res .sum::(); let (selected, mut proofs) = f .directory - .select_node_bundle(&f.node, &proposal, &f.lease, Limits::default(), NOW) + .select_node_bundle(&f.node, &proposal, &f.lease, Limits::default(), now) .await .unwrap(); f.node = selected; @@ -90,7 +91,13 @@ async fn dense_history_retains_215_exact_commands_before_checkpoint_and_cold_res expected.sort(); assert_eq!(outcomes, expected); f.directory - .checkpoint_bundle_cell(&f.node, &cell.authority, &proof, Limits::default(), NOW) + .checkpoint_bundle_cell( + &f.node, + &cell.authority, + &proof, + Limits::default(), + f.node.advertisement().issued_at_ms(), + ) .await .unwrap(); } diff --git a/crates/cellule-runtime/src/node/bundle/tests/lifecycle.rs b/crates/cellule-runtime/src/node/bundle/tests/lifecycle.rs index 8455379f..c11c9d36 100644 --- a/crates/cellule-runtime/src/node/bundle/tests/lifecycle.rs +++ b/crates/cellule-runtime/src/node/bundle/tests/lifecycle.rs @@ -177,6 +177,58 @@ async fn checkpoint_releases_locators_and_the_next_range_continues_exactly() { assert_eq!(proofs[0].locator_count(), frames.len()); } +#[tokio::test] +async fn closed_dense_history_cannot_withdraw_before_the_exact_root_checkpoint() { + let mut f = Fixture::new().await; + let mut cell = f.cell(4).await; + let (_, frames, assigned) = f.append(&mut cell, 2); + let prepared = f + .directory + .prepare_node_bundle(&f.node, &frames, &[assigned], NOW) + .await + .unwrap(); + let (selected, proofs) = f + .directory + .select_node_bundle(&f.node, &prepared, &f.lease, Limits::default(), NOW) + .await + .unwrap(); + let pin = cell.control.value().bundle_binding.unwrap(); + let issued = f + .gate + .close_cell_issuance( + Fixture::scope(&cell), + cell.control.value().ltx_root().unwrap(), + ) + .unwrap(); + let closing = f + .directory + .begin_bundle_close(&selected, pin, issued, NOW) + .await + .unwrap(); + let closed = f + .directory + .finish_bundle_close(&closing, pin, issued, NOW) + .await + .unwrap(); + assert!( + matches!( + f.directory.withdraw(&closed, NOW).await, + Err(Error::PendingPublication) + ), + "Closed rows still retain a dense native suffix until its exact root checkpoint" + ); + f.publisher(&cell) + .materialize_bundle(&proofs[0]) + .await + .unwrap(); + let checkpoint = f + .directory + .checkpoint_bundle_cell(&closed, &cell.authority, &proofs[0], Limits::default(), NOW) + .await + .unwrap(); + f.directory.withdraw(&checkpoint, NOW).await.unwrap(); +} + #[tokio::test] async fn quiet_binding_must_close_and_checkpoint_before_session_withdrawal() { let mut f = Fixture::new().await; diff --git a/crates/cellule-runtime/src/node/bundle/tests/mod.rs b/crates/cellule-runtime/src/node/bundle/tests/mod.rs index 353a42b7..dd0d75bd 100644 --- a/crates/cellule-runtime/src/node/bundle/tests/mod.rs +++ b/crates/cellule-runtime/src/node/bundle/tests/mod.rs @@ -101,6 +101,28 @@ impl Fixture { scratch: tempfile::tempdir().unwrap(), } } + async fn heartbeat(&mut self) -> i64 { + // Density tests deliberately perform hundreds of selections. Renew + // only after canonical heartbeat CAS, preserving the original head + // and production lease duration even on a busy verification host. + let mut next = self.node.advertisement().clone(); + next.issued_at_ms += 1; + next.expires_at_ms += 1; + next.progress += 1; + let key = SigningKey::from_bytes(&[10; 32]); + next.signature = key.sign(&next.signing_bytes().unwrap()).to_bytes(); + let now = next.issued_at_ms; + let refreshed = self.directory.refresh(&self.node, next, now).await.unwrap(); + assert_eq!( + refreshed.advertisement().bundle_head(), + self.node.advertisement().bundle_head() + ); + self.lease + .renew(now, refreshed.advertisement().expires_at_ms()) + .unwrap(); + self.node = refreshed; + now + } async fn cell(&mut self, byte: u8) -> Cell { self.cell_for_application(byte, [9; 16]).await } diff --git a/crates/cellule-runtime/src/node/bundle/tests/ranges.rs b/crates/cellule-runtime/src/node/bundle/tests/ranges.rs index 8e7a186a..01dbcafe 100644 --- a/crates/cellule-runtime/src/node/bundle/tests/ranges.rs +++ b/crates/cellule-runtime/src/node/bundle/tests/ranges.rs @@ -262,15 +262,16 @@ async fn locator_pressure_refuses_selection_without_dropping_the_last_proof() { let mut f = Fixture::new().await; let mut cell = f.cell(4).await; for commit in 2..=(MAX_LOCATORS as u64 + 1) { + let now = f.heartbeat().await; let (_, frames, assigned) = f.append(&mut cell, commit); let prepared = f .directory - .prepare_node_bundle(&f.node, &frames, &[assigned], NOW) + .prepare_node_bundle(&f.node, &frames, &[assigned], now) .await .unwrap(); f.node = f .directory - .select_node_bundle(&f.node, &prepared, &f.lease, Limits::default(), NOW) + .select_node_bundle(&f.node, &prepared, &f.lease, Limits::default(), now) .await .unwrap() .0; @@ -284,7 +285,12 @@ async fn locator_pressure_refuses_selection_without_dropping_the_last_proof() { let (_, frames, assigned) = f.append(&mut cell, MAX_LOCATORS as u64 + 2); assert!(matches!( f.directory - .prepare_node_bundle(&f.node, &frames, &[assigned], NOW) + .prepare_node_bundle( + &f.node, + &frames, + &[assigned], + f.node.advertisement().issued_at_ms(), + ) .await, Err(Error::PendingPublication) )); diff --git a/crates/cellule-runtime/src/node/bundle/tests/recovery/cohort.rs b/crates/cellule-runtime/src/node/bundle/tests/recovery/cohort.rs new file mode 100644 index 00000000..31a4e134 --- /dev/null +++ b/crates/cellule-runtime/src/node/bundle/tests/recovery/cohort.rs @@ -0,0 +1,185 @@ +use super::*; +use std::sync::atomic::Ordering; + +#[tokio::test] +async fn full_recovery_stages_65_real_cell_checkpoints_before_one_terminal_node_cas() { + let faults = Arc::new(super::super::faults::ReplyFault::default()); + let mut f = Fixture::with_store(faults.clone()).await; + enroll(&mut f).await; + let mut cells = Vec::new(); + let mut prefix = Vec::new(); + let mut assignments = Vec::new(); + for byte in 1..=65 { + let mut cell = f.cell(byte).await; + let (_, frames, assigned) = f.append(&mut cell, 2); + assert_eq!( + frames.len(), + 1, + "the small-image fixture has one native frame per command" + ); + prefix.extend(frames); + assignments.push(assigned); + cells.push(cell); + } + for (frames, assigned) in prefix.chunks(64).zip(assignments.chunks(64)) { + let prepared = f + .directory + .prepare_node_bundle(&f.node, frames, assigned, NOW) + .await + .unwrap(); + f.node = f + .directory + .select_node_bundle(&f.node, &prepared, &f.lease, Limits::default(), NOW) + .await + .unwrap() + .0; + } + let through = f + .node + .advertisement() + .bundle_head() + .unwrap() + .selected_through(); + f.node = f + .directory + .advance_log_coverage(&f.node, through, NOW) + .await + .unwrap(); + let mut suffix = Vec::new(); + let mut fleet = Vec::new(); + for cell in &mut cells { + let (_, frames, assigned) = f.append(cell, 3); + suffix.extend(frames); + fleet.push(assigned); + } + let dirs = [tempfile::tempdir().unwrap(), tempfile::tempdir().unwrap()]; + let followers = Arc::new(Followers( + dirs.iter() + .enumerate() + .map(|(i, dir)| { + ( + NodeId::from_bytes([i as u8 + 2; 16]), + FollowerStore::open( + dir.path().to_owned(), + Limits::default(), + cellule_ltx::DiskBudget::new(1 << 30), + ) + .unwrap(), + ) + }) + .collect(), + )); + for member in [2, 3] { + let id = NodeId::from_bytes([member; 16]); + for frames in prefix.chunks(64) { + followers + .append( + id, + AppendRequest { + leader_session: SessionId::from_bytes([1; 16]), + log_epoch: EPOCH, + frames: frames.iter().map(|frame| frame.encoded().clone()).collect(), + covered_through: 0, + }, + ) + .await + .unwrap(); + } + let mut receipt = None; + for frames in suffix.chunks(64) { + receipt = Some( + followers + .append( + id, + AppendRequest { + leader_session: SessionId::from_bytes([1; 16]), + log_epoch: EPOCH, + frames: frames.iter().map(|frame| frame.encoded().clone()).collect(), + covered_through: through, + }, + ) + .await + .unwrap(), + ); + } + let receipt = receipt.unwrap(); + assert!(receipt.base_sequence > through); + f.gate.acknowledge(id, receipt.durable_through).unwrap(); + } + f.gate.activate_fleet().unwrap(); + for assigned in fleet { + assert_eq!( + f.gate.prove(assigned.ticket()).await.unwrap().source(), + DurabilitySource::Fleet + ); + } + let fenced = fence(&f).await; + f.lease.fence(); + let transport: Arc = followers; + let recovery = NodeLogRecovery::from_fenced(transport, &fenced, Limits::default()).unwrap(); + let manifests = RecoveryManifestStore::new(f.layout.clone(), Limits::default()) + .with_recovery_scratch(f.scratch.path().to_owned()); + let inventory = cells + .iter() + .map(|cell| RecoveryCell { + application: ApplicationId::from_bytes(*cell.authority.layout().application_id()), + authority: cell.authority.clone(), + observed: cell.control.clone(), + }) + .collect(); + faults.node_updates.store(0, Ordering::SeqCst); + faults.coverage_puts.store(0, Ordering::SeqCst); + let completed = RecoveryCoordinator::new(recovery, manifests) + .recover_and_seal(&f.directory, fenced, inventory, NOW + 31_000) + .await + .unwrap(); + assert_eq!(completed.controls.len(), 65); + assert_eq!( + faults.coverage_puts.load(Ordering::SeqCst), + 2, + "65 roots stage in bounded 64-root cohorts" + ); + assert_eq!( + faults.node_updates.load(Ordering::SeqCst), + 2, + "one complete terminal catalog CAS, then one log seal" + ); + for cell in &cells { + let current = cell + .authority + .load(cell.control.value().cell) + .await + .unwrap() + .unwrap(); + assert!(current.value().recovery.is_none()); + assert_eq!(current.value().root.as_ref().unwrap().commit_sequence, 3); + let cold = CellReplica::new( + cell.authority.layout().clone(), + *current.value().cell.as_bytes(), + *current.value().incarnation.as_bytes(), + Limits::default(), + ) + .unwrap(); + let destination = f.scratch.path().join(format!( + "cohort-cold-{}.sqlite", + current.value().cell.as_bytes()[0] + )); + cold.open_root(¤t.value().ltx_root().unwrap()) + .await + .unwrap() + .restore(&destination) + .await + .unwrap(); + let restored = cellule_ltx::rusqlite::Connection::open(destination).unwrap(); + for command in [2, 3] { + let result: String = restored + .query_row( + "SELECT result FROM outcomes WHERE request=?1", + [format!("request-{command}")], + |row| row.get(0), + ) + .unwrap(); + assert_eq!(result, format!("result-{command}")); + } + } +} diff --git a/crates/cellule-runtime/src/node/bundle/tests/recovery.rs b/crates/cellule-runtime/src/node/bundle/tests/recovery/mod.rs similarity index 51% rename from crates/cellule-runtime/src/node/bundle/tests/recovery.rs rename to crates/cellule-runtime/src/node/bundle/tests/recovery/mod.rs index 66bce130..0ec0a463 100644 --- a/crates/cellule-runtime/src/node/bundle/tests/recovery.rs +++ b/crates/cellule-runtime/src/node/bundle/tests/recovery/mod.rs @@ -7,6 +7,7 @@ use crate::node::log_transport::{ }; use crate::recovery::manifest::RecoveryManifestStore; use futures_util::future::BoxFuture; +mod cohort; struct Followers(Vec<(NodeId, FollowerStore)>); impl Followers { @@ -122,12 +123,60 @@ async fn fence(f: &Fixture) -> crate::node::FencedNodeSession { enum Scenario { Pruned, + Complete, + PartialRoot, + LostRootReply, + LostCatalogReply, + ExpiredClaim, + SealInterrupted, Overlap, SelectedOnly, Conflict, MissingCell, } +impl Scenario { + fn completes(&self) -> bool { + matches!( + self, + Self::Complete + | Self::PartialRoot + | Self::LostRootReply + | Self::LostCatalogReply + | Self::ExpiredClaim + | Self::SealInterrupted + ) + } +} + +#[tokio::test] +async fn recovery_resumes_closed_origin_inventory_after_transfer_and_interrupted_log_seal() { + verify_recovery(Scenario::SealInterrupted).await; +} + +#[tokio::test] +async fn recovery_reconciles_lost_original_root_cas_reply() { + verify_recovery(Scenario::LostRootReply).await; +} +#[tokio::test] +async fn recovery_reconciles_lost_terminal_tombstone_cas_reply() { + verify_recovery(Scenario::LostCatalogReply).await; +} +#[tokio::test] +async fn expired_claim_cannot_materialize_or_close_original_bound_cells() { + verify_recovery(Scenario::ExpiredClaim).await; +} + +#[tokio::test] +async fn recovery_materializes_full_fleet_suffix_and_closes_quiet_binding_before_transfer() { + verify_recovery(Scenario::Complete).await; +} + +#[tokio::test] +async fn recovery_retries_after_one_root_cas_without_replacing_the_original_manifest() { + verify_recovery(Scenario::PartialRoot).await; +} + #[tokio::test] async fn recovery_joins_selected_only_cell_and_pruned_prefix_with_prior_fleet_ack() { verify_recovery(Scenario::Pruned).await; @@ -150,10 +199,16 @@ async fn recovery_rejects_omitted_selected_only_cell_without_attachment() { } async fn verify_recovery(scenario: Scenario) { - let mut f = Fixture::new().await; + let faults = Arc::new(super::faults::ReplyFault::default()); + let mut f = Fixture::with_store(faults.clone()).await; enroll(&mut f).await; let mut a = f.cell(4).await; let mut b = f.cell(5).await; + let quiet = if scenario.completes() { + Some(f.cell(6).await) + } else { + None + }; let (_, mut prefix, arange) = f.append(&mut a, 2); let (_, bframes, brange) = f.append(&mut b, 2); prefix.extend(bframes); @@ -270,7 +325,8 @@ async fn verify_recovery(scenario: Scenario) { let fenced = fence(&f).await; f.lease.fence(); let transport: Arc = followers; - let recovery = NodeLogRecovery::from_fenced(transport, &fenced, Limits::default()).unwrap(); + let recovery = + NodeLogRecovery::from_fenced(transport.clone(), &fenced, Limits::default()).unwrap(); let sealed = recovery.ensure_sealed_bounded().await.unwrap(); let expected_frames = if coverage == 0 { prefix.len() + suffix.len() @@ -289,14 +345,15 @@ async fn verify_recovery(scenario: Scenario) { .await .unwrap(); assert_eq!( - inventory.len(), - 2, + inventory.controls.len(), + if quiet.is_some() { 3 } else { 2 }, "complete binding discovery includes selected-only Cells" ); let manifests = RecoveryManifestStore::new(f.layout.clone(), Limits::default()) .with_recovery_scratch(f.scratch.path().to_owned()); let cells = [&a, &b] .into_iter() + .chain(quiet.as_ref()) .filter(|cell| { !matches!(scenario, Scenario::MissingCell) || cell.control.value().cell != b.control.value().cell @@ -307,7 +364,7 @@ async fn verify_recovery(scenario: Scenario) { observed: cell.control.clone(), }) .collect(); - let coordinator = RecoveryCoordinator::new(recovery, manifests.clone()); + let mut coordinator = RecoveryCoordinator::new(recovery, manifests.clone()); let result = coordinator .recover_sealed(fenced.clone(), cells, sealed) .await; @@ -338,6 +395,7 @@ async fn verify_recovery(scenario: Scenario) { 2, "selected-only Cells also require durable recovery" ); + let mut expected_roots = std::collections::BTreeMap::new(); for cell in [&a, &b] { let control = controls .iter() @@ -356,6 +414,7 @@ async fn verify_recovery(scenario: Scenario) { .prepare_recovered_overlay(&overlay, 1) .await .unwrap(); + expected_roots.insert(*cell.control.value().cell.as_bytes(), prepared.root()); let path = f.scratch.path().join(format!( "recovered-{}.sqlite", cell.control.value().cell.as_bytes()[0] @@ -384,6 +443,254 @@ async fn verify_recovery(scenario: Scenario) { assert_eq!(result, format!("result-{command}")); } } + if scenario.completes() { + let mut cells = Vec::new(); + for cell in [&a, &b].into_iter().chain(quiet.as_ref()) { + cells.push(RecoveryCell { + application: ApplicationId::from_bytes(*cell.authority.layout().application_id()), + authority: cell.authority.clone(), + observed: cell + .authority + .load(cell.control.value().cell) + .await + .unwrap() + .unwrap(), + }); + } + if matches!(scenario, Scenario::ExpiredClaim) { + assert!( + coordinator + .recover_and_seal( + &f.directory, + fenced.clone(), + cells, + fenced.claim_expires_at_ms() + 1 + ) + .await + .is_err() + ); + for cell in [&a, &b] { + let current = cell + .authority + .load(cell.control.value().cell) + .await + .unwrap() + .unwrap(); + assert_eq!(current.value().root, cell.control.value().root); + assert!(current.value().recovery.is_some()); + } + return; + } + if matches!(scenario, Scenario::PartialRoot) { + let original_manifest = controls[0] + .value() + .recovery + .as_ref() + .unwrap() + .manifest_digest; + faults.mode.store(6, std::sync::atomic::Ordering::SeqCst); + assert!( + coordinator + .recover_and_seal(&f.directory, fenced.clone(), cells.clone(), NOW + 31_000) + .await + .is_err() + ); + let advanced = a + .authority + .load(a.control.value().cell) + .await + .unwrap() + .unwrap(); + assert!(advanced.value().recovery.is_none()); + assert_eq!(advanced.value().root.as_ref().unwrap().commit_sequence, 3); + let pending = b + .authority + .load(b.control.value().cell) + .await + .unwrap() + .unwrap(); + assert_eq!( + pending.value().recovery.as_ref().unwrap().manifest_digest, + original_manifest + ); + assert!(matches!( + a.authority + .transition( + &advanced, + advanced + .value() + .takeover(Owner { + session: fenced.claimant(), + endpoint: "https://successor.internal:8081".into(), + }) + .unwrap(), + Transition::Takeover + ) + .await, + Err(Error::PendingPublication) + )); + faults.mode.store(0, std::sync::atomic::Ordering::SeqCst); + for cell in &mut cells { + cell.observed = cell + .authority + .load(cell.observed.value().cell) + .await + .unwrap() + .unwrap(); + } + } + if matches!(scenario, Scenario::LostRootReply) { + faults.mode.store(8, std::sync::atomic::Ordering::SeqCst); + } else if matches!(scenario, Scenario::LostCatalogReply) { + faults.mode.store(2, std::sync::atomic::Ordering::SeqCst); + } + let mut completion_fence = fenced.clone(); + if matches!(scenario, Scenario::SealInterrupted) { + faults.mode.store(4, std::sync::atomic::Ordering::SeqCst); + assert!( + coordinator + .recover_and_seal(&f.directory, fenced.clone(), cells, NOW + 31_000) + .await + .is_err() + ); + faults.mode.store(0, std::sync::atomic::Ordering::SeqCst); + let original = a + .authority + .load(a.control.value().cell) + .await + .unwrap() + .unwrap(); + a.authority + .transition( + &original, + original + .value() + .takeover(Owner { + session: fenced.claimant(), + endpoint: "https://successor.internal:8081".into(), + }) + .unwrap(), + Transition::Takeover, + ) + .await + .unwrap(); + completion_fence = f + .directory + .claim_expired(fenced.session(), fenced.claimant(), NOW + 31_001) + .await + .unwrap(); + let resumed = NodeLogRecovery::from_fenced( + transport.clone(), + &completion_fence, + Limits::default(), + ) + .unwrap(); + let sealed = resumed.ensure_sealed_bounded().await.unwrap(); + let catalog = crate::cell::catalog::CellCatalog::new( + f.layout.clone(), + crate::TenantId::from_bytes([9; 16]), + ); + let discovered = crate::node::log_recovery::recoverable_cells_from_scopes_with_summary( + &catalog, + &a.authority, + fenced.session(), + &sealed.scopes(Limits::default()).unwrap(), + 4, + ) + .await + .unwrap(); + assert!(discovered.cells.is_empty()); + assert_eq!( + discovered.summary.control_reads, 0, + "closed generations use authenticated original roots, even after transfer" + ); + cells = discovered.cells; + coordinator = RecoveryCoordinator::new(resumed, manifests.clone()); + } + let completed = coordinator + .recover_and_seal(&f.directory, completion_fence, cells, NOW + 31_002) + .await + .unwrap(); + assert_eq!( + completed.controls.len(), + if matches!(scenario, Scenario::SealInterrupted) { + 0 + } else { + 3 + } + ); + for cell in [&a, &b].into_iter().chain(quiet.as_ref()) { + let current = cell + .authority + .load(cell.control.value().cell) + .await + .unwrap() + .unwrap(); + assert!(current.value().recovery.is_none()); + let expected = if current.value().cell == a.control.value().cell { + 3 + } else if current.value().cell == b.control.value().cell { + 2 + } else { + 1 + }; + assert_eq!( + current.value().root.as_ref().unwrap().commit_sequence, + expected + ); + let expected_root = expected_roots + .get(current.value().cell.as_bytes()) + .copied() + .unwrap_or_else(|| cell.control.value().ltx_root().unwrap()); + assert_eq!(current.value().ltx_root(), Some(expected_root)); + let cold = CellReplica::new( + cell.authority.layout().clone(), + *current.value().cell.as_bytes(), + *current.value().incarnation.as_bytes(), + Limits::default(), + ) + .unwrap(); + let destination = f.scratch.path().join(format!( + "cold-terminal-{}.sqlite", + current.value().cell.as_bytes()[0] + )); + cold.open_root(&expected_root) + .await + .unwrap() + .restore(&destination) + .await + .unwrap(); + let restored = cellule_ltx::rusqlite::Connection::open(destination).unwrap(); + for command in 2..=expected { + let actual: String = restored + .query_row( + "SELECT result FROM outcomes WHERE request=?1", + [format!("request-{command}")], + |row| row.get(0), + ) + .unwrap(); + assert_eq!(actual, format!("result-{command}")); + } + if current.value().owner.as_ref().unwrap().session == completed.takeover.claimant() { + assert!(matches!(scenario, Scenario::SealInterrupted)); + continue; + } + let successor = current + .value() + .takeover(Owner { + session: completed.takeover.claimant(), + endpoint: "https://successor.internal:8081".into(), + }) + .unwrap(); + let transferred = cell + .authority + .transition(¤t, successor, Transition::Takeover) + .await + .unwrap(); + assert_eq!(transferred.value().root, current.value().root); + } + return; + } assert!( matches!( coordinator diff --git a/crates/cellule-runtime/src/node/log_recovery/mod.rs b/crates/cellule-runtime/src/node/log_recovery/mod.rs index a70493a5..356af605 100644 --- a/crates/cellule-runtime/src/node/log_recovery/mod.rs +++ b/crates/cellule-runtime/src/node/log_recovery/mod.rs @@ -232,6 +232,7 @@ pub struct NodeLogRecovery { } /// One dead-session Cell control that may need a recovered tail attached. +#[derive(Clone)] pub struct RecoveryCell { /// Application the Cell belongs to. pub application: ApplicationId, @@ -247,11 +248,12 @@ pub struct RecoveryCoordinator { manifests: RecoveryManifestStore, } -/// Completed dead-session recovery with every overlay pinned before log seal. +/// Completed dead-session recovery with every required suffix retained before log seal. pub struct CompletedNodeRecovery { /// Proof that the dead session's node log is sealed. pub sealed: SealedNodeLog, - /// Recovered controls with their overlays attached. + /// Original bound controls after materialization, or unbound controls with + /// exact overlays attached. Already closed generations need no control CAS. pub controls: Vec, /// Proof that the predecessor cannot add newer durable Cell state. pub takeover: NodeTakeoverProof, @@ -265,6 +267,11 @@ pub struct RecoveryCoordinatorResult { pub publication: crate::recovery::manifest::RecoveryPublicationSummary, } +struct RecoveryAttempt { + result: RecoveryCoordinatorResult, + drain: Option, +} + /// Scans one application catalog for published Cells owned by a dead session. pub async fn recoverable_cells( catalog: &CellCatalog, @@ -351,10 +358,20 @@ pub async fn recoverable_cells_from_scopes_with_summary( type Scope = ([u8; 16], u64); let mut scopes = BTreeMap::<[u8; 32], Scope>::new(); let application = *catalog.application().as_bytes(); + let inventory = + crate::node::bundle::recovery::inventory_for_owner(authority.layout(), owner).await?; for scope in frame_scopes { if scope.leader_session != *owner.as_bytes() || scope.application != application { return Err(Error::Node("recovery frame application or owner differs")); } + if inventory.closed.contains(&( + scope.application, + scope.cell, + scope.incarnation, + scope.cell_epoch, + )) { + continue; + } let key = (scope.incarnation, scope.cell_epoch); if let Some(existing) = scopes.get(&scope.cell) { if *existing != key { @@ -374,9 +391,7 @@ pub async fn recoverable_cells_from_scopes_with_summary( // Follower scopes omit Cells whose whole suffix is selected in origin. // Complete authenticated binding discovery is mandatory even for an empty // follower witness; it never grants takeover authority. - for (binding_application, binding) in - crate::node::bundle::recovery::inventory_for_owner(authority.layout(), owner).await? - { + for (binding_application, binding) in inventory.controls { if binding_application.as_bytes() != &application { continue; } @@ -516,6 +531,15 @@ impl RecoveryCoordinator { cells: Vec, sealed: SealedSession, ) -> Result { + Ok(self.recover_attempt(fenced, cells, sealed).await?.result) + } + + async fn recover_attempt( + &self, + fenced: FencedNodeSession, + cells: Vec, + sealed: SealedSession, + ) -> Result { self.recovery.validate_fence(&fenced)?; let mut bases = Vec::with_capacity(cells.len()); for cell in &cells { @@ -547,27 +571,16 @@ impl RecoveryCoordinator { { return Err(Error::Fenced); } - let inventory = - crate::node::bundle::recovery::fenced_inventory(self.manifests.layout(), &fenced) - .await?; - for (application, binding) in &inventory { - let cell = cells - .iter() - .find(|cell| { - cell.application == *application && cell.observed.value().cell == binding.cell - }) - .ok_or(Error::Node( - "selected bundle Cell is absent from recovery inventory", - ))?; - let current = cell.observed.value(); - if current.bundle_binding != binding.bundle_binding - || current.epoch != binding.epoch - || current.incarnation != binding.incarnation - || current.code != binding.code - || current.schema != binding.schema - { - return Err(Error::Fenced); - } + let mut drain = crate::node::bundle::recovery::BundleRecoveryDrain::open( + self.manifests.layout(), + &fenced, + &cells, + sealed.durable_through, + self.recovery.limits, + ) + .await?; + if let Some(drain) = &drain { + drain.add_closed_bases(&mut bases)?; } let scratch = self.manifests.recovery_scratch_directory(); let mut builder = StreamingRecovery::new( @@ -603,13 +616,23 @@ impl RecoveryCoordinator { let frame = frame?; first_sequence.get_or_insert(frame.scope().node_sequence); last_sequence = Some(frame.scope().node_sequence); + let scope = frame.scope(); + let position = frame.segment().position(); builder.push(frame)?; + if let Some(drain) = &mut drain { + drain.observe(scope, position)?; + } } } else { for frame in sealed.frames { first_sequence.get_or_insert(frame.scope().node_sequence); last_sequence = Some(frame.scope().node_sequence); + let scope = frame.scope(); + let position = frame.segment().position(); builder.push(frame)?; + if let Some(drain) = &mut drain { + drain.observe(scope, position)?; + } } } if first_sequence.is_some_and(|first| first != required_first) @@ -619,14 +642,25 @@ impl RecoveryCoordinator { } let tails = builder.finish()?; if tails.is_empty() { - return Ok(RecoveryCoordinatorResult { - controls: Vec::new(), - publication: crate::recovery::manifest::RecoveryPublicationSummary::default(), + return Ok(RecoveryAttempt { + result: RecoveryCoordinatorResult { + controls: Vec::new(), + publication: crate::recovery::manifest::RecoveryPublicationSummary::default(), + }, + drain, }); } let pinned = self .manifests - .pin_with_summary(sealed.leader_session, sealed.log_epoch, tails) + .pin_reconciled( + sealed.leader_session, + sealed.log_epoch, + tails, + &cells + .iter() + .filter_map(|cell| cell.observed.value().recovery.clone()) + .collect::>(), + ) .await?; if pinned.cells.len() > cells.len() { return Err(Error::Node("recovery manifest exceeds Cell inventory")); @@ -678,13 +712,18 @@ impl RecoveryCoordinator { }; attached.push(versioned); } - Ok(RecoveryCoordinatorResult { - controls: attached, - publication: pinned.summary, + Ok(RecoveryAttempt { + result: RecoveryCoordinatorResult { + controls: attached, + publication: pinned.summary, + }, + drain, }) } /// Pins every recovered Cell and then atomically seals the claimed node log. + /// Bound original writers first materialize the complete verified suffix + /// and checkpoint every binding, including quiet Cells, under this claim. pub async fn recover_and_seal( &self, directory: &NodeDirectory, @@ -692,8 +731,27 @@ impl RecoveryCoordinator { cells: Vec, now_ms: i64, ) -> Result { - let controls = self.recover(fenced.clone(), cells).await?; - self.finish(directory, fenced, controls, now_ms).await + let started = std::time::Instant::now(); + self.recovery.validate_fence(&fenced)?; + let witness = self.recovery.ensure_sealed_bounded().await?; + let attempt = self.recover_attempt(fenced.clone(), cells, witness).await?; + let Some(drain) = attempt.drain else { + return self + .finish(directory, fenced, attempt.result.controls, now_ms) + .await; + }; + let manifest = pinned_manifest(&attempt.result.controls, &fenced, self.recovery.log_epoch)?; + let (fresh, controls) = drain + .complete(directory, &self.manifests, &fenced, now_ms, started) + .await?; + let now = crate::node::bundle::recovery::logical_now(now_ms, started)?; + let sealed = directory.seal_recovery(&fresh, manifest, now).await?; + let takeover = NodeTakeoverProof::after_recovery(&fenced, &sealed)?; + Ok(CompletedNodeRecovery { + sealed, + controls, + takeover, + }) } /// Seals one still-current claim after all recovered controls were pinned. @@ -705,28 +763,7 @@ impl RecoveryCoordinator { now_ms: i64, ) -> Result { self.recovery.validate_fence(&fenced)?; - let mut manifest = None::; - for control in &controls { - let recovery = control - .value() - .recovery - .as_ref() - .ok_or(Error::Control("recovered Cell has no pinned overlay"))?; - if recovery.leader_session != fenced.session() - || recovery.log_epoch != self.recovery.log_epoch - { - return Err(Error::Control("recovered Cell overlay scope differs")); - } - match manifest { - None => manifest = Some(recovery.manifest_digest), - Some(current) if current == recovery.manifest_digest => {} - Some(_) => { - return Err(Error::Control( - "recovered session produced multiple manifests", - )); - } - } - } + let manifest = pinned_manifest(&controls, &fenced, self.recovery.log_epoch)?; let sealed = directory.seal_recovery(&fenced, manifest, now_ms).await?; let takeover = NodeTakeoverProof::after_recovery(&fenced, &sealed)?; Ok(CompletedNodeRecovery { @@ -737,6 +774,34 @@ impl RecoveryCoordinator { } } +fn pinned_manifest( + controls: &[VersionedControl], + fenced: &FencedNodeSession, + epoch: u64, +) -> Result> { + let mut manifest = None; + for control in controls { + let recovery = control + .value() + .recovery + .as_ref() + .ok_or(Error::Control("recovered Cell has no pinned overlay"))?; + if recovery.leader_session != fenced.session() || recovery.log_epoch != epoch { + return Err(Error::Control("recovered Cell overlay scope differs")); + } + match manifest { + None => manifest = Some(recovery.manifest_digest), + Some(current) if current == recovery.manifest_digest => {} + Some(_) => { + return Err(Error::Control( + "recovered session produced multiple manifests", + )); + } + } + } + Ok(manifest) +} + impl NodeLogRecovery { pub(crate) fn new( transport: Arc, diff --git a/crates/cellule-runtime/src/publication/lineage.rs b/crates/cellule-runtime/src/publication/lineage.rs index 0c045f5c..355d527a 100644 --- a/crates/cellule-runtime/src/publication/lineage.rs +++ b/crates/cellule-runtime/src/publication/lineage.rs @@ -3,7 +3,7 @@ use super::*; use cellule_ltx::{RootPreparation, RootPreparationFuture, RootPreparationMetadata}; use std::sync::{Arc, Mutex}; -pub(super) type LineageConfirmation = Arc>>; +pub(crate) type LineageConfirmation = Arc>>; struct RootLineageRecorder { authority: CellAuthority, @@ -25,7 +25,7 @@ impl RootPreparationMetadata for RootLineageRecorder { } } -pub(super) fn replica( +pub(crate) fn replica( replica: cellule_ltx::CellReplica, authority: &CellAuthority, ) -> (cellule_ltx::CellReplica, LineageConfirmation) { @@ -37,7 +37,7 @@ pub(super) fn replica( (replica, confirmed) } -pub(super) fn error(source: cellule_ltx::LtxError) -> Error { +pub(crate) fn error(source: cellule_ltx::LtxError) -> Error { match source { cellule_ltx::LtxError::RootPreparation { source } => match source.downcast::() { Ok(source) => *source, diff --git a/crates/cellule-runtime/src/publication/mod.rs b/crates/cellule-runtime/src/publication/mod.rs index d6c56900..daa9b8ac 100644 --- a/crates/cellule-runtime/src/publication/mod.rs +++ b/crates/cellule-runtime/src/publication/mod.rs @@ -10,7 +10,7 @@ use crate::node::log_shipper::NodeLogSubmission; use crate::retry::{Backoff, retry_hint, retryable_storage_error}; use crate::{Error, Result}; -mod lineage; +pub(crate) mod lineage; mod shared; pub(crate) use shared::{PublicationPermit, SharedPublication}; diff --git a/crates/cellule-runtime/src/recovery/manifest/mod.rs b/crates/cellule-runtime/src/recovery/manifest/mod.rs index 50c8599c..ee984b11 100644 --- a/crates/cellule-runtime/src/recovery/manifest/mod.rs +++ b/crates/cellule-runtime/src/recovery/manifest/mod.rs @@ -17,6 +17,7 @@ const MAX_MANIFEST_CELLS: usize = 4_096; const MULTIPART_BYTES: usize = 8 << 20; mod inventory; +mod reconcile; pub use inventory::RecoveryManifestInventory; /// One control-ready pointer returned after bundle and manifest publication. @@ -191,6 +192,11 @@ pub struct RecoveryManifestStore { } impl RecoveryManifestStore { + pub(crate) fn for_application(&self, application: ApplicationId) -> Self { + let mut store = self.clone(); + store.layout = store.layout.for_application(*application.as_bytes()); + store + } pub(crate) fn layout(&self) -> &CellStorageLayout { &self.layout } @@ -483,16 +489,17 @@ impl RecoveryManifestStore { .await?; let limits = self.limits; let decoded = tokio::task::spawn_blocking(move || { - cellule_ltx::bundle::Bundle::decode_temp_file_with_digest( + let bundle = cellule_ltx::bundle::Bundle::decode_temp_file_with_digest( temporary, row.bundle_digest, limits, - ) + )?; + Ok::<_, cellule_ltx::LtxError>((bundle, disk_reservation)) }) .await .map_err(cellule_ltx::LtxError::from)?; - let bundle = match decoded { - Ok(bundle) => bundle, + let (bundle, disk_reservation) = match decoded { + Ok(result) => result, Err(cellule_ltx::LtxError::ChecksumMismatch) => { return Err(Error::Node("recovery bundle digest differs")); } diff --git a/crates/cellule-runtime/src/recovery/manifest/reconcile.rs b/crates/cellule-runtime/src/recovery/manifest/reconcile.rs new file mode 100644 index 00000000..af1f7ba9 --- /dev/null +++ b/crates/cellule-runtime/src/recovery/manifest/reconcile.rs @@ -0,0 +1,67 @@ +//! Reuse the exact original manifest after a partial materialization. +use super::*; + +impl RecoveryManifestStore { + pub(crate) async fn pin_reconciled( + &self, + leader: SessionId, + epoch: u64, + tails: Vec, + existing: &[RecoveryOverlayRef], + ) -> Result { + let Some(first) = existing.first() else { + return self.pin_with_summary(leader, epoch, tails).await; + }; + if existing.iter().any(|reference| { + reference.leader_session != leader + || reference.log_epoch != epoch + || reference.manifest_digest != first.manifest_digest + }) { + return Err(Error::Control( + "recovered session produced multiple manifests", + )); + } + let manifest = self + .load_manifest_body(leader, epoch, first.manifest_digest) + .await?; + let mut cells = Vec::with_capacity(tails.len()); + for tail in tails { + let base = tail.overlay.predecessor(); + let row = manifest + .cells + .iter() + .find(|row| { + row.scope() + == ( + tail.application, + base.cell, + base.incarnation, + tail.cell_epoch, + ) + }) + .ok_or(Error::Node("original recovery manifest omits pending Cell"))?; + if row.predecessor != base + || row.first_node_sequence != tail.first_node_sequence + || row.last_node_sequence != tail.last_node_sequence + || row.final_position != tail.overlay.final_position() + || row.final_commit_sequence != tail.overlay.final_commit_sequence() + || row.bundle_digest != tail.overlay.bundle().digest() + { + return Err(Error::Node( + "original recovery manifest differs from verified tail", + )); + } + // The newly rebuilt bytes authenticate this exact original object. + // Already materialized siblings remain in that manifest; a subset + // retry cannot replace the pending controls with a different digest. + cells.push(row.pinned(leader, epoch, first.manifest_digest)); + } + Ok(PinnedRecoveryCells { + cells, + summary: RecoveryPublicationSummary { + object_reads: 1, + ..Default::default() + }, + }) + } +} diff --git a/docs/bundle-coverage-implementation.md b/docs/bundle-coverage-implementation.md index 79a0aba2..bd1ee6df 100644 --- a/docs/bundle-coverage-implementation.md +++ b/docs/bundle-coverage-implementation.md @@ -48,9 +48,16 @@ inventory includes selected-only Cells; identical native overlap is skipped, while conflicting bytes and an omitted Cell reject before attachment. A bound control may retain only its original recovery session/epoch after canonical fencing. Its pin remains until complete terminal materialization/checkpoint; -the lower-level node seal enforces that obligation too. Failed-boot materializer -scheduling, terminal catalog CAS and provisional enrollment resolution remain -unfinished. This does not enable ordinary bundle responses. +the lower-level node seal enforces that obligation too. `recover_and_seal` now +materializes bound overlays under the fenced original binding, retaining normal +lineage and exact root CAS. It verifies every original root, including quiet +Cells, stages at most 64 checkpoints per immutable upload, then selects the +complete terminal catalog with one fresh-claim node CAS before sealing the log. +Only exact lost replies reconcile. Partial-root retries retain the original +manifest; verified closed roots allow resumption after transfer or interrupted +log seal. Provisional enrollment reconciliation, admitted host scheduling and +full lifecycle qualification remain unfinished. Ordinary bundle responses remain +disabled. The original SQL/capture/submission jobs must join before `close_cell_issuance`. Its ordered gate prevents late assignment from consuming a node sequence. The @@ -178,8 +185,26 @@ file-backed follower fsync, verifying a pruned prefix with a prior Fleet ACK, identical overlap, selected-only recovery, conflicting overlap and omitted inventory. Separate tests cover a 2,000-scope manifest codec and retention of scratch admission when a cancelled waiter leaves cache work running. The scope -codec is a metadata fixture, not 2,000 writers or throughput evidence. Failed-boot -root materialization and terminal catalog closure remain required. +codec is a metadata fixture, not 2,000 writers or throughput evidence. + +The terminal recovery regressions cold-restore selected and prior Fleet outcomes +before allowing Cell transfer. They cover a partial root CAS, lost root/catalog +replies, expired claims and resumption after terminal catalog selection when one +Cell has transferred but log seal was interrupted. A **65-real-Cell** case crosses +the cohort boundary: **two immutable catalog uploads, one complete terminal node +CAS, then one log seal**. This excludes per-Cell roots, recovery manifest uploads, +enrollment and steady-state application work; it is not TPS or full M4 evidence. +The existing 215-command/four-PUT test remains passing. + +The terminal-recovery snapshot passed all twelve contributor checks: **1,920 +top-level workspace cases passed, 38 ignored; 60 local LTX cases passed**. +This count excludes four nested subprocess executions included in earlier +totals. Default-parallel verification first expired four unchanged lease/grant +fixtures; a four-worker control run passed the identical runtime source. The +long density fixtures now renew only after authoritative heartbeat CAS, retaining +the same production lease duration and catalog head. The final isolated suite +uses four test workers; qualification concurrency and thresholds are unchanged. +These checks do not measure application TPS or complete lifecycle qualification. The following counts describe the earlier inline-index snapshot, not the latest application TPS: @@ -267,10 +292,10 @@ Remaining work before responses can use bundle proof: 2. Build on the authenticated index: bound admitted maintenance inventory, increase checkpoint density with retained-byte accounting, and measure the complete materialization/checkpoint/collection cost. -3. Complete failed-boot materialization and terminal catalog selection after - combined prefix/follower reconstruction, including prior Fleet ACKs and - interrupted provisional enrollments. Current departure and canonical seal - guards refuse unfinished closure rather than advancing a replacement writer. +3. Finish interrupted provisional enrollment reconciliation and admitted + failed-boot orchestration around combined reconstruction, exact materialization + and atomic terminal selection. Departure and canonical seal guards still + refuse any unfinished original binding before replacement admission. 4. Implement complete cross-Cell reference inventory, pins and grace-qualified collection. Preserve original catalog and base dependencies throughout. 5. Run the unchanged all-ACK cold recovery, paired Docker throughput/latency, From 07f271298d149e025879c09c526ab215145a0758 Mon Sep 17 00:00:00 2001 From: forhappy Date: Wed, 7 Oct 2026 20:57:57 -0700 Subject: [PATCH 023/102] Reconcile fenced provisional bundle reservations --- .../docs/failover-and-followers.md | 10 +- .../docs/write-performance-design.md | 10 +- .../src/node/bundle/recovery/drain.rs | 85 +++- .../src/node/bundle/recovery/mod.rs | 66 ++- .../src/node/bundle/tests/faults.rs | 13 + .../src/node/bundle/tests/index/cohort.rs | 16 +- .../src/node/bundle/tests/mod.rs | 11 +- .../src/node/bundle/tests/ranges.rs | 55 ++- .../src/node/bundle/tests/recovery/cohort.rs | 16 +- .../src/node/bundle/tests/recovery/mod.rs | 1 + .../node/bundle/tests/recovery/provisional.rs | 389 ++++++++++++++++++ .../src/node/log_recovery/mod.rs | 22 +- docs/bundle-coverage-implementation.md | 45 +- 13 files changed, 676 insertions(+), 63 deletions(-) create mode 100644 crates/cellule-runtime/src/node/bundle/tests/recovery/provisional.rs diff --git a/crates/cellule-runtime/docs/failover-and-followers.md b/crates/cellule-runtime/docs/failover-and-followers.md index b5077160..ae643133 100644 --- a/crates/cellule-runtime/docs/failover-and-followers.md +++ b/crates/cellule-runtime/docs/failover-and-followers.md @@ -1589,8 +1589,14 @@ the full witness and terminal endpoint must still agree. Closed original roots also supply verified recovery bases after transfer, so interruption between terminal catalog CAS and log seal does not reload a -successor as the failed writer. Provisional enrollment resolution and admitted -host scheduling remain unfinished; ordinary bundle ACKs remain disabled. The +successor as the failed writer. Quiet provisional reservations are classified +against fresh Cell authority and their exact verified original base. Pinned +reservations require the original Cell in recovery; unpinned reservations close +without changing that Cell. An already-authorized late pin CAS cannot reopen +the closed node catalog. A pre-activation native frame rejects before attachment +or terminal selection and retains the unresolved obligation. Discovery counts +those authority reads even when no affected Cell catalog pages are needed. +Admitted host scheduling remains unfinished; ordinary bundle ACKs remain disabled. The manifest reader admits up to 4,096 scopes under the existing 2 MiB byte ceiling; the 2,000-scope codec test is format evidence, not node performance qualification. diff --git a/crates/cellule-runtime/docs/write-performance-design.md b/crates/cellule-runtime/docs/write-performance-design.md index 673dd87b..a2e85dab 100644 --- a/crates/cellule-runtime/docs/write-performance-design.md +++ b/crates/cellule-runtime/docs/write-performance-design.md @@ -112,6 +112,10 @@ now materializes original bound roots through ordinary lineage and exact Cell CAS, then stages bounded catalog cohorts and selects their complete terminal inventory with one recovery-claim node CAS. Quiet Cells participate. Exact manifest reuse supports partial-root retries, and closed original roots support -resumption after transfer or interrupted log seal. Provisional enrollment -resolution, admitted host scheduling and full qualification remain required -before enabling the canonical coverage frontier or actor ACKs. +resumption after transfer or interrupted log seal. Quiet provisional reservations +now close under the same fenced claim after checking current Cell authority and +verifying the original base. An unpinned reservation needs no Cell CAS; a late pin CAS +cannot reopen the closed catalog. Pre-activation native frames remain an explicit +unresolved obligation and cannot be discarded. Admitted host scheduling and full +qualification remain required before enabling the canonical coverage frontier +or actor ACKs. diff --git a/crates/cellule-runtime/src/node/bundle/recovery/drain.rs b/crates/cellule-runtime/src/node/bundle/recovery/drain.rs index e9aace8a..4f0f4414 100644 --- a/crates/cellule-runtime/src/node/bundle/recovery/drain.rs +++ b/crates/cellule-runtime/src/node/bundle/recovery/drain.rs @@ -19,6 +19,7 @@ pub(crate) struct BundleRecoveryDrain { original: NodeBundleHead, scopes: BTreeMap, closed: BTreeMap, + unpinned: BTreeMap, durable_through: u64, } @@ -40,6 +41,7 @@ impl BundleRecoveryDrain { } let mut scopes = BTreeMap::new(); let mut closed = BTreeMap::new(); + let mut unpinned = BTreeMap::new(); for binding in fenced_bindings(layout, fenced).await? { if binding.phase == BindingPhase::Closed { if binding.terminal @@ -65,10 +67,22 @@ impl BundleRecoveryDrain { ); continue; } - // Interrupted enrollment still needs explicit reconciliation. It - // must never disappear merely because no follower row names it. if binding.phase == BindingPhase::Provisional { - return Err(Error::PendingPublication); + let authority = + CellAuthority::new(layout.for_application(*binding.application.as_bytes())); + let current = authority + .load(binding.control.cell) + .await? + .ok_or(Error::Fenced)?; + let pinned = provisional_pinned(&binding, current.value())?; + verify_base(layout, &binding, limits).await?; + if !pinned { + // Preserve the original reservation root even if the + // unpinned Cell has independently released or moved. A + // late original pin CAS can only select this exact base. + unpinned.insert(generation(&binding), binding); + continue; + } } let cell = cells .iter() @@ -109,6 +123,7 @@ impl BundleRecoveryDrain { original, scopes, closed, + unpinned, durable_through, })) } @@ -118,6 +133,14 @@ impl BundleRecoveryDrain { native: cellule_ltx::NodeFrameScope, position: cellule_ltx::Position, ) -> Result<()> { + if self.unpinned.contains_key(&( + native.application, + native.cell, + native.incarnation, + native.cell_epoch, + )) { + return Err(Error::Node("provisional Cell issued before activation")); + } if let Some(closed) = self.closed.get(&( native.application, native.cell, @@ -136,6 +159,9 @@ impl BundleRecoveryDrain { .scopes .get_mut(&(native.application, native.cell)) .ok_or(Error::Node("sealed bundle frame has no original binding"))?; + if scope.binding.phase == BindingPhase::Provisional { + return Err(Error::Node("provisional Cell issued before activation")); + } if native.incarnation != *scope.binding.control.incarnation.as_bytes() || native.cell_epoch != scope.binding.control.epoch { @@ -147,11 +173,11 @@ impl BundleRecoveryDrain { Ok(()) } - pub(crate) fn add_closed_bases( + pub(crate) fn add_archived_bases( &self, bases: &mut Vec, ) -> Result<()> { - for (key, binding) in &self.closed { + for (key, binding) in self.closed.iter().chain(&self.unpinned) { if bases.iter().any(|base| { ( base.application, @@ -174,6 +200,15 @@ impl BundleRecoveryDrain { Ok(()) } + pub(crate) fn is_provisional(&self, cell: &RecoveryCell) -> bool { + self.scopes + .get(&( + *cell.application.as_bytes(), + *cell.observed.value().cell.as_bytes(), + )) + .is_some_and(|scope| scope.binding.phase == BindingPhase::Provisional) + } + /// Materialize every original bound Cell before changing any catalog row. /// A late failure retains original pins and immutable recovery pointers. pub(crate) async fn complete( @@ -261,42 +296,52 @@ impl BundleRecoveryDrain { verify_base(manifests.layout(), &materialized, manifests.limits()).await?; controls.push(current); } - let entries = self + let mut entries = self .scopes .into_values() .zip(controls.iter()) + .map(|(scope, control)| (scope.binding, scope.terminal, control.value().clone())) .collect::>(); + entries.extend(self.unpinned.into_values().map(|binding| { + let terminal = ( + binding.selected_sequence, + binding.selected_commit, + binding.selected_position, + ); + let control = binding.control.clone(); + (binding, terminal, control) + })); let mut staged = self.original; for cohort in entries.chunks(MAX_FRAMES) { let head = staged; let keys = cohort .iter() - .map(|(scope, _)| { + .map(|(binding, _, _)| { ( - *scope.binding.application.as_bytes(), - *scope.binding.control.cell.as_bytes(), + *binding.application.as_bytes(), + *binding.control.cell.as_bytes(), ) }) .collect(); let mut catalog = store::load_catalog_cells(manifests.layout(), fenced.session(), head, &keys) .await?; - for (scope, control) in cohort { - let pin = scope.binding.control.bundle_binding.ok_or(Error::Fenced)?; + for (original, terminal, control) in cohort { + let pin = original.control.bundle_binding.ok_or(Error::Fenced)?; let binding = catalog.binding_mut(pin.digest)?; - if !same_scope(&binding.control, control.value()) - || binding.phase == BindingPhase::Provisional - || binding.selected_sequence > scope.terminal.0 + if !same_scope(&binding.control, control) + || binding.phase != original.phase + || binding.selected_sequence > terminal.0 { return Err(Error::Fenced); } - binding.control = control.value().clone(); + binding.control = control.clone(); binding.phase = BindingPhase::Closed; - binding.terminal = Some(scope.terminal); - binding.selected_sequence = scope.terminal.0; - binding.selected_commit = scope.terminal.1; - binding.selected_position = scope.terminal.2; - binding.first_commit = scope.terminal.1; + binding.terminal = Some(*terminal); + binding.selected_sequence = terminal.0; + binding.selected_commit = terminal.1; + binding.selected_position = terminal.2; + binding.first_commit = terminal.1; binding.locators.clear(); } // Every original root was verified above, including quiet Cells. diff --git a/crates/cellule-runtime/src/node/bundle/recovery/mod.rs b/crates/cellule-runtime/src/node/bundle/recovery/mod.rs index 61bef8c4..499d886e 100644 --- a/crates/cellule-runtime/src/node/bundle/recovery/mod.rs +++ b/crates/cellule-runtime/src/node/bundle/recovery/mod.rs @@ -33,6 +33,8 @@ async fn record(layout: &cellule_ltx::CellStorageLayout, session: SessionId) -> pub(crate) struct OwnerBundleInventory { pub(crate) controls: Vec<(crate::ApplicationId, Control)>, pub(crate) closed: std::collections::BTreeSet, + pub(crate) unpinned: std::collections::BTreeSet, + pub(crate) control_reads: u64, } pub(crate) async fn inventory_for_owner( @@ -66,13 +68,21 @@ async fn inventory_at( let mut inventory = OwnerBundleInventory::default(); for binding in index::binding_inventory(layout, session, head).await? { if binding.phase == BindingPhase::Closed { - inventory.closed.insert(( - *binding.application.as_bytes(), - *binding.control.cell.as_bytes(), - *binding.control.incarnation.as_bytes(), - binding.control.epoch, - )); + inventory.closed.insert(generation(&binding)); } else { + if binding.phase == BindingPhase::Provisional { + let authority = + CellAuthority::new(layout.for_application(*binding.application.as_bytes())); + let current = authority + .load(binding.control.cell) + .await? + .ok_or(Error::Fenced)?; + inventory.control_reads += 1; + if !provisional_pinned(&binding, current.value())? { + inventory.unpinned.insert(generation(&binding)); + continue; + } + } inventory .controls .push((binding.application, binding.control)); @@ -81,6 +91,50 @@ async fn inventory_at( Ok(inventory) } +fn generation(binding: &Binding) -> GenerationKey { + ( + *binding.application.as_bytes(), + *binding.control.cell.as_bytes(), + *binding.control.incarnation.as_bytes(), + binding.control.epoch, + ) +} + +fn provisional_pinned(binding: &Binding, current: &Control) -> Result { + let base = binding + .control + .ltx_root() + .ok_or(Error::PendingPublication)?; + if binding.phase != BindingPhase::Provisional + || binding.control.state != crate::control::ControlState::Serving + || binding.control.bundle_binding.is_none() + || binding.control.recovery.is_some() + || binding.terminal.is_some() + || !binding.locators.is_empty() + || binding.selected_sequence != 0 + || binding.first_commit != base.commit_sequence + || binding.selected_commit != base.commit_sequence + || binding.selected_position != base.position + || current.cell != binding.control.cell + || binding.control.owner.as_ref().map(|owner| owner.session) + != binding.control.bundle_binding.map(|pin| pin.session) + { + return Err(Error::PendingPublication); + } + if current.bundle_binding != binding.control.bundle_binding { + return Ok(false); + } + // Renewal may have changed bookkeeping, but admission remains closed until + // activation. No mutation, migration or different base may be inferred. + let mut expected = binding.control.clone(); + expected.revision = current.revision; + expected.progress = current.progress; + if expected != *current { + return Err(Error::PendingPublication); + } + Ok(true) +} + async fn fenced_bindings( layout: &cellule_ltx::CellStorageLayout, fenced: &FencedNodeSession, diff --git a/crates/cellule-runtime/src/node/bundle/tests/faults.rs b/crates/cellule-runtime/src/node/bundle/tests/faults.rs index a4c43356..c5066546 100644 --- a/crates/cellule-runtime/src/node/bundle/tests/faults.rs +++ b/crates/cellule-runtime/src/node/bundle/tests/faults.rs @@ -12,6 +12,8 @@ pub(super) struct ReplyFault { pub(super) mode: AtomicU8, pub(super) node_updates: AtomicUsize, pub(super) coverage_puts: AtomicUsize, + pub(super) pin_started: tokio::sync::Notify, + pub(super) pin_resume: tokio::sync::Notify, } impl std::fmt::Display for ReplyFault { fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { @@ -32,6 +34,17 @@ impl ObjectStore for ReplyFault { opts: PutOptions, ) -> object_store::Result { let mode = self.mode.load(Ordering::SeqCst); + if mode == 9 + && path.as_ref().ends_with("/control.json") + && matches!(opts.mode, object_store::PutMode::Update(_)) + && self + .mode + .compare_exchange(9, 0, Ordering::SeqCst, Ordering::SeqCst) + .is_ok() + { + self.pin_started.notify_one(); + self.pin_resume.notified().await; + } if path.as_ref().ends_with("/control.json") && matches!(opts.mode, object_store::PutMode::Update(_)) { diff --git a/crates/cellule-runtime/src/node/bundle/tests/index/cohort.rs b/crates/cellule-runtime/src/node/bundle/tests/index/cohort.rs index 00c2f42c..fe1f06aa 100644 --- a/crates/cellule-runtime/src/node/bundle/tests/index/cohort.rs +++ b/crates/cellule-runtime/src/node/bundle/tests/index/cohort.rs @@ -6,6 +6,7 @@ async fn checkpoint_cohort_uses_two_puts_and_preserves_a_hot_cell_suffix() { let mut cells = Vec::new(); for number in 4..(4 + MAX_FRAMES as u8) { cells.push(f.cell(number).await); + f.heartbeat().await; } let mut frames = Vec::new(); let mut assigned = Vec::new(); @@ -16,12 +17,23 @@ async fn checkpoint_cohort_uses_two_puts_and_preserves_a_hot_cell_suffix() { } let proposal = f .directory - .prepare_node_bundle(&f.node, &frames, &assigned, NOW) + .prepare_node_bundle( + &f.node, + &frames, + &assigned, + f.node.advertisement().issued_at_ms(), + ) .await .unwrap(); let (node, proofs) = f .directory - .select_node_bundle(&f.node, &proposal, &f.lease, Limits::default(), NOW) + .select_node_bundle( + &f.node, + &proposal, + &f.lease, + Limits::default(), + f.node.advertisement().issued_at_ms(), + ) .await .unwrap(); f.node = node; diff --git a/crates/cellule-runtime/src/node/bundle/tests/mod.rs b/crates/cellule-runtime/src/node/bundle/tests/mod.rs index dd0d75bd..736c20c3 100644 --- a/crates/cellule-runtime/src/node/bundle/tests/mod.rs +++ b/crates/cellule-runtime/src/node/bundle/tests/mod.rs @@ -28,6 +28,7 @@ struct Fixture { lease: NodeLeaseGuard, gate: DurabilityGate, scratch: tempfile::TempDir, + capture_host: cellule_ltx::Host, } struct Cell { db: Db, @@ -40,6 +41,12 @@ impl Fixture { Self::with_store(Arc::new(InMemory::new())).await } async fn with_store(store: Arc) -> Self { + Self::with_capture_host(store, cellule_ltx::Host::default()).await + } + async fn with_capture_host( + store: Arc, + capture_host: cellule_ltx::Host, + ) -> Self { let count = Arc::new(CountingObjectStore::new(store)); let layout = CellStorageLayout::new( Store::new(count.clone()), @@ -99,6 +106,7 @@ impl Fixture { lease, gate, scratch: tempfile::tempdir().unwrap(), + capture_host, } } async fn heartbeat(&mut self) -> i64 { @@ -149,12 +157,13 @@ impl Fixture { let layout = self.layout.for_application(application); let cell = CellId::from_bytes([byte; 32]); let incarnation = IncarnationId::from_bytes([byte + 10; 16]); - let mut db = Db::open( + let mut db = Db::open_with_host( &self.scratch.path().join(format!( "{byte}-{}.sqlite", crate::identity::encode_hex(&application) )), Limits::default(), + self.capture_host.clone(), ) .unwrap(); db.transaction(|tx| tx.execute_batch("CREATE TABLE outcomes(request TEXT PRIMARY KEY, result TEXT); INSERT INTO outcomes VALUES ('seed','original')")).unwrap(); diff --git a/crates/cellule-runtime/src/node/bundle/tests/ranges.rs b/crates/cellule-runtime/src/node/bundle/tests/ranges.rs index 01dbcafe..fbc18233 100644 --- a/crates/cellule-runtime/src/node/bundle/tests/ranges.rs +++ b/crates/cellule-runtime/src/node/bundle/tests/ranges.rs @@ -257,13 +257,46 @@ async fn corrupt_origin_dependency_and_overlapping_ranges_fail_closed() { ); } +struct CheckpointClock(std::sync::atomic::AtomicBool); + +impl cellule_ltx::environment::Clock for CheckpointClock { + fn unix_millis(&self) -> i64 { + NOW + } + + fn file_age(&self, _: &std::path::Path) -> std::io::Result { + // Exercise the real time-based checkpoint once, independently of the + // verification host's speed. All native cuts still enter the bundle. + Ok(if self.0.swap(false, std::sync::atomic::Ordering::SeqCst) { + std::time::Duration::from_secs(61) + } else { + std::time::Duration::ZERO + }) + } +} + #[tokio::test] async fn locator_pressure_refuses_selection_without_dropping_the_last_proof() { - let mut f = Fixture::new().await; + let clock = Arc::new(CheckpointClock(std::sync::atomic::AtomicBool::new(false))); + let mut f = Fixture::with_capture_host( + Arc::new(InMemory::new()), + cellule_ltx::Host::default().with_clock(clock.clone()), + ) + .await; let mut cell = f.cell(4).await; - for commit in 2..=(MAX_LOCATORS as u64 + 1) { + let mut selected_frames = 0; + let mut last_commit = 1; + while selected_frames < MAX_LOCATORS { + let commit = last_commit + 1; let now = f.heartbeat().await; + if commit == 128 { + clock.0.store(true, std::sync::atomic::Ordering::SeqCst); + } let (_, frames, assigned) = f.append(&mut cell, commit); + if commit == 128 { + assert_eq!(frames.len(), 2, "checkpoint contributes its native cut"); + } + assert!(selected_frames + frames.len() <= MAX_LOCATORS); let prepared = f .directory .prepare_node_bundle(&f.node, &frames, &[assigned], now) @@ -275,6 +308,8 @@ async fn locator_pressure_refuses_selection_without_dropping_the_last_proof() { .await .unwrap() .0; + selected_frames += frames.len(); + last_commit = commit; } let last = f .directory @@ -282,7 +317,8 @@ async fn locator_pressure_refuses_selection_without_dropping_the_last_proof() { .await .unwrap(); assert_eq!(last.locator_count(), MAX_LOCATORS); - let (_, frames, assigned) = f.append(&mut cell, MAX_LOCATORS as u64 + 2); + assert_eq!(last_commit, MAX_LOCATORS as u64); + let (_, frames, assigned) = f.append(&mut cell, last_commit + 1); assert!(matches!( f.directory .prepare_node_bundle( @@ -294,7 +330,7 @@ async fn locator_pressure_refuses_selection_without_dropping_the_last_proof() { .await, Err(Error::PendingPublication) )); - assert_eq!(last.commit_sequence(), MAX_LOCATORS as u64 + 1); + assert_eq!(last.commit_sequence(), last_commit); assert_eq!( f.directory .load_bundle_coverage(&cell.authority, &cell.control, Limits::default()) @@ -306,8 +342,9 @@ async fn locator_pressure_refuses_selection_without_dropping_the_last_proof() { f.count.reset(); let root = f.publisher(&cell).materialize_bundle(&last).await.unwrap(); eprintln!( - "bounded suffix materialization: locators={} puts={}", + "bounded suffix materialization: locators={} commands={} puts={}", MAX_LOCATORS, + last_commit - 1, f.count.put_requests() ); assert_eq!( @@ -324,14 +361,10 @@ async fn locator_pressure_refuses_selection_without_dropping_the_last_proof() { .await .unwrap(); let db = rusqlite::Connection::open(restored).unwrap(); - let count: usize = db + let count: u64 = db .query_row("SELECT count(*) FROM outcomes", [], |row| row.get(0)) .unwrap(); - assert_eq!( - count, - MAX_LOCATORS + 1, - "the later unselected command is absent" - ); + assert_eq!(count, last_commit, "the later unselected command is absent"); } #[tokio::test] diff --git a/crates/cellule-runtime/src/node/bundle/tests/recovery/cohort.rs b/crates/cellule-runtime/src/node/bundle/tests/recovery/cohort.rs index 31a4e134..0864377a 100644 --- a/crates/cellule-runtime/src/node/bundle/tests/recovery/cohort.rs +++ b/crates/cellule-runtime/src/node/bundle/tests/recovery/cohort.rs @@ -11,6 +11,7 @@ async fn full_recovery_stages_65_real_cell_checkpoints_before_one_terminal_node_ let mut assignments = Vec::new(); for byte in 1..=65 { let mut cell = f.cell(byte).await; + f.heartbeat().await; let (_, frames, assigned) = f.append(&mut cell, 2); assert_eq!( frames.len(), @@ -24,12 +25,23 @@ async fn full_recovery_stages_65_real_cell_checkpoints_before_one_terminal_node_ for (frames, assigned) in prefix.chunks(64).zip(assignments.chunks(64)) { let prepared = f .directory - .prepare_node_bundle(&f.node, frames, assigned, NOW) + .prepare_node_bundle( + &f.node, + frames, + assigned, + f.node.advertisement().issued_at_ms(), + ) .await .unwrap(); f.node = f .directory - .select_node_bundle(&f.node, &prepared, &f.lease, Limits::default(), NOW) + .select_node_bundle( + &f.node, + &prepared, + &f.lease, + Limits::default(), + f.node.advertisement().issued_at_ms(), + ) .await .unwrap() .0; diff --git a/crates/cellule-runtime/src/node/bundle/tests/recovery/mod.rs b/crates/cellule-runtime/src/node/bundle/tests/recovery/mod.rs index 0ec0a463..aafd312d 100644 --- a/crates/cellule-runtime/src/node/bundle/tests/recovery/mod.rs +++ b/crates/cellule-runtime/src/node/bundle/tests/recovery/mod.rs @@ -8,6 +8,7 @@ use crate::node::log_transport::{ use crate::recovery::manifest::RecoveryManifestStore; use futures_util::future::BoxFuture; mod cohort; +mod provisional; struct Followers(Vec<(NodeId, FollowerStore)>); impl Followers { diff --git a/crates/cellule-runtime/src/node/bundle/tests/recovery/provisional.rs b/crates/cellule-runtime/src/node/bundle/tests/recovery/provisional.rs new file mode 100644 index 00000000..d0fccf1c --- /dev/null +++ b/crates/cellule-runtime/src/node/bundle/tests/recovery/provisional.rs @@ -0,0 +1,389 @@ +use super::*; +use std::sync::atomic::Ordering; + +async fn interrupted(mode: u8) -> (Fixture, Cell, Arc) { + let faults = Arc::new(super::super::faults::ReplyFault::default()); + let mut f = Fixture::with_store(faults.clone()).await; + let mut node = f.node.advertisement().clone(); + node.log = Some( + NodeLogStatus::open( + node.node, + EPOCH, + vec![NodeId::from_bytes([2; 16]), NodeId::from_bytes([3; 16])], + ) + .unwrap(), + ); + node.generation += 1; + f.node = f + .directory + .update_advertisement(&f.node, node, NOW) + .await + .unwrap(); + let mut cell = f.unbound_cell_for_application(4, [9; 16]).await; + faults.mode.store(mode, Ordering::SeqCst); + assert!( + f.directory + .bind_bundle_cell(&f.node, &cell.authority, &cell.control, NOW) + .await + .is_err() + ); + faults.mode.store(0, Ordering::SeqCst); + f.node = f + .directory + .load(SessionId::from_bytes([1; 16]), NOW) + .await + .unwrap() + .unwrap(); + cell.control = cell + .authority + .load(cell.control.value().cell) + .await + .unwrap() + .unwrap(); + assert_eq!(cell.control.value().bundle_binding.is_some(), mode == 4); + let followers = Arc::new(Followers( + [2, 3] + .into_iter() + .map(|number| { + ( + NodeId::from_bytes([number; 16]), + FollowerStore::open( + f.scratch.path().join(format!("provisional-{number}")), + Limits::default(), + cellule_ltx::DiskBudget::new(1 << 30), + ) + .unwrap(), + ) + }) + .collect(), + )); + (f, cell, followers) +} + +#[tokio::test] +async fn failed_boot_closes_an_unpinned_quiet_reservation_without_a_cell_cas() { + quiet(3, false).await; +} + +#[tokio::test] +async fn failed_boot_closes_a_pinned_quiet_reservation_before_departure() { + quiet(4, false).await; +} + +#[tokio::test] +async fn failed_boot_closes_an_unpinned_reservation_after_the_cell_released() { + quiet(3, true).await; +} + +async fn quiet(mode: u8, released: bool) { + let (f, mut cell, followers) = interrupted(mode).await; + if released { + let next = cell.control.value().release().unwrap(); + cell.control = cell + .authority + .transition(&cell.control, next, Transition::Release) + .await + .unwrap(); + } + let original = cell.control.clone(); + let reservation = load_catalog( + &f.layout, + SessionId::from_bytes([1; 16]), + f.node.advertisement().bundle_head().unwrap(), + ) + .await + .unwrap() + .bindings + .remove(0); + if mode == 4 { + assert!(matches!( + cell.authority + .transition( + &original, + original.value().release().unwrap(), + Transition::Release + ) + .await, + Err(Error::PendingPublication) + )); + } + let fenced = fence(&f).await; + f.lease.fence(); + if mode == 3 && !released { + assert!( + cell.authority + .transition(&original, reservation.control, Transition::BindBundle) + .await + .is_err(), + "a new enrollment authorization cannot cross the original node fence" + ); + } + let discovery = super::super::super::recovery::inventory_for_owner( + &f.layout, + SessionId::from_bytes([1; 16]), + ) + .await + .unwrap(); + assert_eq!(discovery.controls.len(), usize::from(mode == 4)); + assert_eq!(discovery.unpinned.len(), usize::from(mode == 3)); + assert_eq!(discovery.control_reads, 1); + if mode == 3 { + let catalog = crate::cell::catalog::CellCatalog::new( + f.layout.clone(), + crate::TenantId::from_bytes([9; 16]), + ); + let discovered = crate::node::log_recovery::recoverable_cells_from_scopes_with_summary( + &catalog, + &cell.authority, + SessionId::from_bytes([1; 16]), + &[], + 1, + ) + .await + .unwrap(); + assert!(discovered.cells.is_empty()); + assert_eq!(discovered.summary.control_reads, 1); + assert_eq!(discovered.summary.catalog_pages, 0); + } + // An unpinned reservation never selected Cell authority. Its original + // catalog root remains an obligation, independently of a later owner. + let inventory = if mode == 4 { + vec![RecoveryCell { + application: ApplicationId::from_bytes([9; 16]), + authority: cell.authority.clone(), + observed: cell.control.clone(), + }] + } else { + Vec::new() + }; + let recovery = NodeLogRecovery::from_fenced(followers, &fenced, Limits::default()).unwrap(); + let manifests = RecoveryManifestStore::new(f.layout.clone(), Limits::default()) + .with_recovery_scratch(f.scratch.path().to_owned()); + f.count.reset(); + let completed = RecoveryCoordinator::new(recovery, manifests) + .recover_and_seal(&f.directory, fenced, inventory, NOW + 31_000) + .await + .unwrap(); + assert_eq!(completed.controls.len(), usize::from(mode == 4)); + assert_eq!( + f.count.put_requests(), + 3, + "catalog upload, terminal CAS and log seal; no Cell root CAS" + ); + let current = cell + .authority + .load(original.value().cell) + .await + .unwrap() + .unwrap(); + assert_eq!(current.value(), original.value()); + let archived = super::super::super::recovery::inventory_for_owner( + &f.layout, + SessionId::from_bytes([1; 16]), + ) + .await + .unwrap(); + assert!(archived.controls.is_empty()); + assert_eq!(archived.closed.len(), 1); + let destination = f.scratch.path().join("provisional-cold.sqlite"); + cell.replica + .open_root(&original.value().ltx_root().unwrap()) + .await + .unwrap() + .restore(&destination) + .await + .unwrap(); + let restored = cellule_ltx::rusqlite::Connection::open(destination).unwrap(); + let result: String = restored + .query_row( + "SELECT result FROM outcomes WHERE request='seed'", + [], + |row| row.get(0), + ) + .unwrap(); + assert_eq!(result, "original"); + if mode == 4 { + cell.authority + .transition( + ¤t, + current.value().release().unwrap(), + Transition::Release, + ) + .await + .unwrap(); + } +} + +#[tokio::test] +async fn provisional_native_frames_cannot_be_discarded_as_quiet_reservations() { + for mode in [3, 4] { + let (mut f, mut cell, followers) = interrupted(mode).await; + let (_, frames, assigned) = f.append(&mut cell, 2); + for member in [2, 3] { + let id = NodeId::from_bytes([member; 16]); + let receipt = followers + .append( + id, + AppendRequest { + leader_session: SessionId::from_bytes([1; 16]), + log_epoch: EPOCH, + frames: frames.iter().map(|frame| frame.encoded().clone()).collect(), + covered_through: 0, + }, + ) + .await + .unwrap(); + f.gate.acknowledge(id, receipt.durable_through).unwrap(); + } + f.gate.activate_fleet().unwrap(); + assert_eq!( + f.gate.prove(assigned.ticket()).await.unwrap().source(), + DurabilitySource::Fleet + ); + f.node = f.directory.activate_log(&f.node, NOW).await.unwrap(); + let fenced = fence(&f).await; + let original_head = fenced.bundle_head(); + f.lease.fence(); + let recovery = NodeLogRecovery::from_fenced(followers, &fenced, Limits::default()).unwrap(); + let manifests = RecoveryManifestStore::new(f.layout.clone(), Limits::default()) + .with_recovery_scratch(f.scratch.path().to_owned()); + f.count.reset(); + let error = RecoveryCoordinator::new(recovery, manifests) + .recover_and_seal( + &f.directory, + fenced.clone(), + vec![RecoveryCell { + application: ApplicationId::from_bytes([9; 16]), + authority: cell.authority.clone(), + observed: cell.control.clone(), + }], + NOW + 31_000, + ) + .await + .err() + .unwrap(); + assert!(matches!( + error, + Error::Node("provisional Cell issued before activation") + )); + assert_eq!( + f.count.put_requests(), + 0, + "a pre-activation frame cannot select terminal state" + ); + let (record, _) = f + .directory + .load_record_at( + &f.layout + .node_path(SessionId::from_bytes([1; 16]).as_bytes()), + ) + .await + .unwrap() + .unwrap(); + let crate::node::directory::NodeRecord::Tombstone(current) = record else { + panic!("failed original boot must remain fenced"); + }; + assert_eq!(current.bundle, original_head); + assert_eq!( + current.log.unwrap().phase(), + crate::node::log_state::NodeLogPhase::Recovering + ); + let current = cell + .authority + .load(cell.control.value().cell) + .await + .unwrap() + .unwrap(); + assert_eq!(current.value(), cell.control.value()); + } +} + +#[tokio::test] +async fn a_pin_cas_already_authorized_before_fencing_cannot_reopen_the_closed_reservation() { + let faults = Arc::new(super::super::faults::ReplyFault::default()); + let mut f = Fixture::with_store(faults.clone()).await; + let mut node = f.node.advertisement().clone(); + node.log = Some( + NodeLogStatus::open( + node.node, + EPOCH, + vec![NodeId::from_bytes([2; 16]), NodeId::from_bytes([3; 16])], + ) + .unwrap(), + ); + node.generation += 1; + f.node = f + .directory + .update_advertisement(&f.node, node, NOW) + .await + .unwrap(); + let cell = f.unbound_cell_for_application(4, [9; 16]).await; + let original_root = cell.control.value().ltx_root().unwrap(); + faults.mode.store(9, Ordering::SeqCst); + let directory = f.directory.clone(); + let authority = cell.authority.clone(); + let control = cell.control.clone(); + let observed = f.node.clone(); + let task = tokio::spawn(async move { + directory + .bind_bundle_cell(&observed, &authority, &control, NOW) + .await + }); + faults.pin_started.notified().await; + f.node = f + .directory + .load(SessionId::from_bytes([1; 16]), NOW) + .await + .unwrap() + .unwrap(); + let fenced = fence(&f).await; + f.lease.fence(); + let followers = Arc::new(Followers( + [2, 3] + .into_iter() + .map(|member| { + ( + NodeId::from_bytes([member; 16]), + FollowerStore::open( + f.scratch.path().join(format!("held-pin-{member}")), + Limits::default(), + cellule_ltx::DiskBudget::new(1 << 30), + ) + .unwrap(), + ) + }) + .collect(), + )); + let recovery = NodeLogRecovery::from_fenced(followers, &fenced, Limits::default()).unwrap(); + let manifests = RecoveryManifestStore::new(f.layout.clone(), Limits::default()) + .with_recovery_scratch(f.scratch.path().to_owned()); + let completed = RecoveryCoordinator::new(recovery, manifests) + .recover_and_seal(&f.directory, fenced, Vec::new(), NOW + 31_000) + .await + .unwrap(); + assert!(completed.controls.is_empty()); + faults.pin_resume.notify_one(); + assert!(task.await.unwrap().is_err()); + let current = cell + .authority + .load(cell.control.value().cell) + .await + .unwrap() + .unwrap(); + assert!(current.value().bundle_binding.is_some()); + assert_eq!(current.value().ltx_root().unwrap(), original_root); + let proof = f + .directory + .load_bundle_coverage(&cell.authority, ¤t, Limits::default()) + .await + .unwrap(); + assert_eq!(proof.locator_count(), 0); + cell.authority + .transition( + ¤t, + current.value().release().unwrap(), + Transition::Release, + ) + .await + .unwrap(); +} diff --git a/crates/cellule-runtime/src/node/log_recovery/mod.rs b/crates/cellule-runtime/src/node/log_recovery/mod.rs index 356af605..e2046937 100644 --- a/crates/cellule-runtime/src/node/log_recovery/mod.rs +++ b/crates/cellule-runtime/src/node/log_recovery/mod.rs @@ -364,12 +364,13 @@ pub async fn recoverable_cells_from_scopes_with_summary( if scope.leader_session != *owner.as_bytes() || scope.application != application { return Err(Error::Node("recovery frame application or owner differs")); } - if inventory.closed.contains(&( + let generation = ( scope.application, scope.cell, scope.incarnation, scope.cell_epoch, - )) { + ); + if inventory.closed.contains(&generation) || inventory.unpinned.contains(&generation) { continue; } let key = (scope.incarnation, scope.cell_epoch); @@ -417,7 +418,10 @@ pub async fn recoverable_cells_from_scopes_with_summary( if scopes.is_empty() { return Ok(RecoverableCellInventory { cells: Vec::new(), - summary: RecoveryInventorySummary::default(), + summary: RecoveryInventorySummary { + control_reads: inventory.control_reads, + ..RecoveryInventorySummary::default() + }, }); } @@ -478,7 +482,9 @@ pub async fn recoverable_cells_from_scopes_with_summary( catalog_shards, catalog_pages, control_reads: u64::try_from(recovered.len()) - .map_err(|_| Error::Capacity("recovery control read count"))?, + .map_err(|_| Error::Capacity("recovery control read count"))? + .checked_add(inventory.control_reads) + .ok_or(Error::Capacity("recovery control read count"))?, }, cells: recovered, }) @@ -580,7 +586,7 @@ impl RecoveryCoordinator { ) .await?; if let Some(drain) = &drain { - drain.add_closed_bases(&mut bases)?; + drain.add_archived_bases(&mut bases)?; } let scratch = self.manifests.recovery_scratch_directory(); let mut builder = StreamingRecovery::new( @@ -592,7 +598,11 @@ impl RecoveryCoordinator { Some(self.recovery.recovery_disk.clone()), )?; for cell in &cells { - if cell.observed.value().bundle_binding.is_some() { + if cell.observed.value().bundle_binding.is_some() + && !drain + .as_ref() + .is_some_and(|drain| drain.is_provisional(cell)) + { builder.seed( crate::node::bundle::recovery::selected_frames( self.manifests.layout(), diff --git a/docs/bundle-coverage-implementation.md b/docs/bundle-coverage-implementation.md index bd1ee6df..bb041aa2 100644 --- a/docs/bundle-coverage-implementation.md +++ b/docs/bundle-coverage-implementation.md @@ -55,9 +55,14 @@ Cells, stages at most 64 checkpoints per immutable upload, then selects the complete terminal catalog with one fresh-claim node CAS before sealing the log. Only exact lost replies reconcile. Partial-root retries retain the original manifest; verified closed roots allow resumption after transfer or interrupted -log seal. Provisional enrollment reconciliation, admitted host scheduling and -full lifecycle qualification remain unfinished. Ordinary bundle responses remain -disabled. +log seal. Quiet provisional enrollment is reconciled against current Cell +authority and its verified original base. An unpinned reservation can close +without changing a Cell that has released or moved. A pin CAS authorized before +fencing may still select that exact original base, but cannot reopen its closed +catalog; departure then uses the ordinary closed-binding guard. Native frames +issued before activation fail recovery before attachment or terminal selection, +retaining the unresolved obligation. Admitted host scheduling and full lifecycle +qualification remain unfinished. Ordinary bundle responses remain disabled. The original SQL/capture/submission jobs must join before `close_cell_issuance`. Its ordered gate prevents late assignment from consuming a node sequence. The @@ -196,16 +201,35 @@ CAS, then one log seal**. This excludes per-Cell roots, recovery manifest upload enrollment and steady-state application work; it is not TPS or full M4 evidence. The existing 215-command/four-PUT test remains passing. -The terminal-recovery snapshot passed all twelve contributor checks: **1,920 +The terminal-recovery snapshot passed all twelve contributor checks: **1,923 top-level workspace cases passed, 38 ignored; 60 local LTX cases passed**. -This count excludes four nested subprocess executions included in earlier -totals. Default-parallel verification first expired four unchanged lease/grant +The earlier 1,920 count incorrectly excluded three distinct compile-fail +doctests as nested executions. Only one actual nested follower subprocess is +excluded from this corrected count. Default-parallel verification first expired four unchanged lease/grant fixtures; a four-worker control run passed the identical runtime source. The long density fixtures now renew only after authoritative heartbeat CAS, retaining the same production lease duration and catalog head. The final isolated suite uses four test workers; qualification concurrency and thresholds are unchanged. These checks do not measure application TPS or complete lifecycle qualification. +The provisional-recovery snapshot passed all twelve contributor checks: +**1,928 top-level workspace cases passed, 38 ignored; 60 local LTX cases passed**. +Five new cases cover quiet unpinned and pinned reservations, a released Cell, +an original pin CAS delayed across fencing, and rejection of pre-activation +issued frames before attachment or terminal selection. Discovery counts its +fresh authority reads even when no Cell catalog pages are scanned. The locator +boundary now forces a production time-based checkpoint through a controlled +file-age input: 255 SQL commands contribute 256 complete native frames, still +materializing with four PUTs and excluding the next unselected command. It does +not equate commands with frames or increase either protocol bound. + +The first complete suite stopped on an unchanged application deadline test +with three admitted reads instead of four. That exact binary passed alone, +and the identical frozen source passed the complete workspace rerun with the +same four test workers and deadlines. The earlier aborted incomplete snapshot, +failed runs, source hashes and count correction remain outside Git. No new +application TPS was measured, and bundle-based responses remain disabled. + The following counts describe the earlier inline-index snapshot, not the latest application TPS: @@ -292,10 +316,11 @@ Remaining work before responses can use bundle proof: 2. Build on the authenticated index: bound admitted maintenance inventory, increase checkpoint density with retained-byte accounting, and measure the complete materialization/checkpoint/collection cost. -3. Finish interrupted provisional enrollment reconciliation and admitted - failed-boot orchestration around combined reconstruction, exact materialization - and atomic terminal selection. Departure and canonical seal guards still - refuse any unfinished original binding before replacement admission. +3. Integrate admitted failed-boot orchestration around combined reconstruction, + provisional reconciliation, exact materialization and atomic terminal + selection. Keep admission closed until enrollment activates. Departure and + canonical seal guards refuse unfinished original bindings before replacement + admission; pre-activation issuance is not a recoverable quiet reservation. 4. Implement complete cross-Cell reference inventory, pins and grace-qualified collection. Preserve original catalog and base dependencies throughout. 5. Run the unchanged all-ACK cold recovery, paired Docker throughput/latency, From 49579857ac4e5b9015ecb88b0a1a78e283154460 Mon Sep 17 00:00:00 2001 From: forhappy Date: Wed, 7 Oct 2026 21:53:23 -0700 Subject: [PATCH 024/102] Select native bundle coverage with one authority CAS --- .../cellule-axum/examples/sql_metrics/mod.rs | 4 +- .../examples/sql_metrics/tests.rs | 8 + .../docs/write-performance-design.md | 8 +- .../src/cell/actor/admission.rs | 2 +- crates/cellule-runtime/src/node/bundle/mod.rs | 10 +- .../cellule-runtime/src/node/bundle/proof.rs | 1 + .../src/node/bundle/selection.rs | 78 ++- .../cellule-runtime/src/node/bundle/store.rs | 45 ++ .../src/node/bundle/tests/coverage.rs | 473 ++++++++++++++++++ .../src/node/bundle/tests/faults.rs | 139 +++++ .../node/bundle/tests/index/compatibility.rs | 2 + .../src/node/bundle/tests/mod.rs | 1 + .../src/node/bundle/tests/recovery/cohort.rs | 9 +- .../src/node/bundle/tests/recovery/mod.rs | 29 +- .../src/node/durability/mod.rs | 15 +- .../src/node/durability/object_coverage.rs | 35 +- .../src/node/durability/tests.rs | 37 ++ crates/cellule-runtime/src/node/lease.rs | 4 + crates/cellule-runtime/src/node/log/mod.rs | 129 ++++- crates/cellule-runtime/src/publication/mod.rs | 8 + .../cellule-runtime/src/publication/tests.rs | 52 ++ docs/bundle-coverage-implementation.md | 28 +- 22 files changed, 1074 insertions(+), 43 deletions(-) create mode 100644 crates/cellule-runtime/src/node/bundle/tests/coverage.rs diff --git a/crates/cellule-axum/examples/sql_metrics/mod.rs b/crates/cellule-axum/examples/sql_metrics/mod.rs index 343fca4d..d5f83a67 100644 --- a/crates/cellule-axum/examples/sql_metrics/mod.rs +++ b/crates/cellule-axum/examples/sql_metrics/mod.rs @@ -32,7 +32,7 @@ pub(super) struct QueryMetrics { response_sources: [AtomicU64; 3], response_elapsed: [Histogram; 3], response_confirmation: Histogram, - proof_wait: [Histogram; 2], + proof_wait: [Histogram; 3], submission_sources: [AtomicU64; 4], log_append_successes: AtomicU64, log_append_failures: AtomicU64, @@ -266,6 +266,7 @@ impl CellTelemetry for QueryMetrics { let index = match source { cellule_runtime::node::log::DurabilitySource::Fleet => 0, cellule_runtime::node::log::DurabilitySource::Object => 1, + cellule_runtime::node::log::DurabilitySource::Bundle => 2, }; self.proof_wait[index].observe(waited); } @@ -413,6 +414,7 @@ impl QueryMetrics { ("response_confirmation", &self.response_confirmation), ("proof_fleet", &self.proof_wait[0]), ("proof_object", &self.proof_wait[1]), + ("proof_bundle", &self.proof_wait[2]), ] { histograms.insert(name.into(), histogram.raw()); } diff --git a/crates/cellule-axum/examples/sql_metrics/tests.rs b/crates/cellule-axum/examples/sql_metrics/tests.rs index fa673096..a7a0e693 100644 --- a/crates/cellule-axum/examples/sql_metrics/tests.rs +++ b/crates/cellule-axum/examples/sql_metrics/tests.rs @@ -33,6 +33,10 @@ fn response_and_proof_timings_keep_ack_latency_separate_from_materialization() { cellule_runtime::node::log::DurabilitySource::Object, Duration::from_millis(50), ); + metrics.durability_proof( + cellule_runtime::node::log::DurabilitySource::Bundle, + Duration::from_millis(3), + ); let sample = metrics.window_snapshot(); assert_eq!( sample["histograms"]["response_fleet"]["nonzero_buckets"], @@ -48,6 +52,10 @@ fn response_and_proof_timings_keep_ack_latency_separate_from_materialization() { ); assert_eq!(sample["response_sources"]["fleet"], 1); assert_eq!(sample["response_sources"]["object"], 0); + assert_eq!( + sample["histograms"]["proof_bundle"]["nonzero_buckets"], + serde_json::json!([[30, 1]]) + ); } #[tokio::test] diff --git a/crates/cellule-runtime/docs/write-performance-design.md b/crates/cellule-runtime/docs/write-performance-design.md index a2e85dab..bf052aae 100644 --- a/crates/cellule-runtime/docs/write-performance-design.md +++ b/crates/cellule-runtime/docs/write-performance-design.md @@ -117,5 +117,9 @@ now close under the same fenced claim after checking current Cell authority and verifying the original base. An unpinned reservation needs no Cell CAS; a late pin CAS cannot reopen the closed catalog. Pre-activation native frames remain an explicit unresolved obligation and cannot be discarded. Admitted host scheduling and full -qualification remain required before enabling the canonical coverage frontier -or actor ACKs. +qualification remain required before enabling actor ACKs. The verified native +selector now advances enrolled coverage and selects its bundle in one node CAS. +Exact local confirmation performs no second CAS and reports a distinct Bundle +proof. Original gate and lease identity remain required; cold proofs grant +reconstruction only. Ordinary actors continue waiting for their root fallback +until command/read/retry visibility and capture release consume that proof. diff --git a/crates/cellule-runtime/src/cell/actor/admission.rs b/crates/cellule-runtime/src/cell/actor/admission.rs index f2737967..a4e0fbcd 100644 --- a/crates/cellule-runtime/src/cell/actor/admission.rs +++ b/crates/cellule-runtime/src/cell/actor/admission.rs @@ -130,7 +130,7 @@ pub(super) fn send_command_reply( let (source, confirmation) = match command.response_proof { Some((DurabilitySource::Fleet, elapsed)) => (CommandResponseSource::Fleet, elapsed), - Some((DurabilitySource::Object, elapsed)) => { + Some((DurabilitySource::Object | DurabilitySource::Bundle, elapsed)) => { (CommandResponseSource::Object, elapsed) } None => (CommandResponseSource::Recorded, std::time::Duration::ZERO), diff --git a/crates/cellule-runtime/src/node/bundle/mod.rs b/crates/cellule-runtime/src/node/bundle/mod.rs index a937b479..26939212 100644 --- a/crates/cellule-runtime/src/node/bundle/mod.rs +++ b/crates/cellule-runtime/src/node/bundle/mod.rs @@ -6,7 +6,9 @@ //! //! The caller owns host admission and the original node lease. This selection //! helper operates on complete captures assigned by the canonical shipper; -//! returned reconstruction proofs do not enable the actor's command response. +//! live selection can confirm the original assigned captures locally without +//! another CAS. Ordinary actor bundle responses remain disabled until their +//! command/read/retry visibility and capture release consume that exact proof. //! //! ```no_run //! use cellule_runtime::node::{NodeDirectory, VersionedNodeAdvertisement}; @@ -45,6 +47,7 @@ mod selection; #[cfg(test)] use proof::checkpoint_prefix; use proof::{verify_base, verify_binding}; +pub(crate) use selection::confirm_selected_coverage; #[cfg(test)] use store::load_catalog; pub(crate) mod store; @@ -138,17 +141,22 @@ pub struct PreparedNodeBundle { catalog: Catalog, body: Bytes, head: NodeBundleHead, + assignments: Vec, } /// Selected, dependency-verified coverage of one exact Cell writer. /// /// The bounded locators retain no frame bodies. Creation is restricted to a /// successful/reconciled canonical node CAS with prior complete range verification. +/// Live selections also retain the original process lease and complete native +/// assignments. Cold reconstruction proofs cannot confirm a live ACK gate. pub struct BundleCoverageProof { pin: BundleBindingRef, binding: Binding, head: NodeBundleHead, session: SessionId, + // Cold reconstruction must never revive the original process's ACK gate. + live: Option, } impl BundleCoverageProof { /// Original Cell authority pin. diff --git a/crates/cellule-runtime/src/node/bundle/proof.rs b/crates/cellule-runtime/src/node/bundle/proof.rs index 2c0f395c..c91cf1ec 100644 --- a/crates/cellule-runtime/src/node/bundle/proof.rs +++ b/crates/cellule-runtime/src/node/bundle/proof.rs @@ -101,6 +101,7 @@ pub(super) async fn load_binding_at( binding, head, session: pin.session, + live: None, }, frames, )) diff --git a/crates/cellule-runtime/src/node/bundle/selection.rs b/crates/cellule-runtime/src/node/bundle/selection.rs index 3272c520..cbc6526f 100644 --- a/crates/cellule-runtime/src/node/bundle/selection.rs +++ b/crates/cellule-runtime/src/node/bundle/selection.rs @@ -3,6 +3,12 @@ use super::*; use crate::node::lease::NodeLeaseGuard; use crate::node::{NodeDirectory, VersionedNodeAdvertisement}; +pub(super) struct LiveBundleCoverage { + node: crate::identity::NodeId, + lease: NodeLeaseGuard, + assignments: Vec, +} + impl NodeDirectory { /// Verifies a contiguous complete native range and uploads one proposal for /// every participating Cell. Neither upload nor this value grants an ACK. @@ -106,7 +112,9 @@ impl NodeDirectory { }); catalog.selected_through = scope.node_sequence; } - self.upload_catalog(Some(head), catalog, frames).await + let mut prepared = self.upload_catalog(Some(head), catalog, frames).await?; + prepared.assignments = assignments.to_vec(); + Ok(prepared) } /// Selects exactly the uploaded catalog/range after I/O, under the original @@ -121,6 +129,9 @@ impl NodeDirectory { now_ms: i64, ) -> Result<(VersionedNodeAdvertisement, Vec)> { lease.check()?; + if prepared.assignments.is_empty() { + return Err(Error::Node("native bundle has no complete assignments")); + } let cells = prepared .catalog .bindings @@ -170,12 +181,75 @@ impl NodeDirectory { binding, head: prepared.head, session: catalog.session, + live: None, }); } lease.check()?; - let selected = self.select_catalog(observed, prepared, now_ms).await?; + let selected = self + .select_native_catalog(observed, prepared, now_ms) + .await?; // Expiry during verification must not revive the original writer. lease.check()?; + // A boot without enrolled native log state supplies reconstruction + // only. Enrollment and the selected coverage frontier are prerequisites + // for waking this process's original durability gate. + if selected.advertisement.log.is_some() { + for proof in &mut proofs { + let scope = crate::node::log::CellLogScope { + application: proof.binding.application, + cell: proof.binding.control.cell, + incarnation: proof.binding.control.incarnation, + cell_epoch: proof.binding.control.epoch, + }; + let assignments = prepared + .assignments + .iter() + .filter(|assignment| assignment.scope() == scope) + .copied() + .collect::>(); + if assignments.is_empty() { + return Err(Error::Node("selected binding lacks complete assignments")); + } + proof.live = Some(LiveBundleCoverage { + node: selected.advertisement.node, + lease: lease.clone(), + assignments, + }); + } + } Ok((selected, proofs)) } } + +pub(crate) fn confirm_selected_coverage( + gate: &crate::node::log::DurabilityGate, + lease: &NodeLeaseGuard, + proofs: &[BundleCoverageProof], +) -> Result { + if proofs.is_empty() { + return Err(Error::Node("empty selected bundle coverage")); + } + lease.check()?; + let (session, node, epoch) = gate.identity()?; + let mut assignments = Vec::new(); + for proof in proofs { + let live = proof + .live + .as_ref() + .ok_or(Error::Node("bundle proof grants reconstruction only"))?; + if live.node != node + || proof.session != session + || proof.head.epoch != epoch + || !live.lease.same_lease(lease) + { + return Err(Error::Fenced); + } + live.lease.check()?; + assignments.extend_from_slice(&live.assignments); + } + // Validate every original capture before waking any sibling. The canonical + // selection already persisted coverage; this is local confirmation only. + let through = gate.confirm_bundle_ranges(&assignments)?; + lease.check()?; + Ok(through) +} diff --git a/crates/cellule-runtime/src/node/bundle/store.rs b/crates/cellule-runtime/src/node/bundle/store.rs index facd613f..c0b9a279 100644 --- a/crates/cellule-runtime/src/node/bundle/store.rs +++ b/crates/cellule-runtime/src/node/bundle/store.rs @@ -38,6 +38,7 @@ impl NodeDirectory { catalog, body, head, + assignments: Vec::new(), }) } @@ -46,6 +47,32 @@ impl NodeDirectory { observed: &VersionedNodeAdvertisement, prepared: &PreparedNodeBundle, now_ms: i64, + ) -> Result { + self.select_catalog_with_coverage(observed, prepared, None, now_ms) + .await + } + + pub(super) async fn select_native_catalog( + &self, + observed: &VersionedNodeAdvertisement, + prepared: &PreparedNodeBundle, + now_ms: i64, + ) -> Result { + self.select_catalog_with_coverage( + observed, + prepared, + Some(prepared.head.selected_through), + now_ms, + ) + .await + } + + async fn select_catalog_with_coverage( + &self, + observed: &VersionedNodeAdvertisement, + prepared: &PreparedNodeBundle, + native_through: Option, + now_ms: i64, ) -> Result { let mut base = observed.clone(); if index::body_digest(&prepared.body)? != prepared.head.digest { @@ -55,6 +82,7 @@ impl NodeDirectory { for _ in 0..4 { self.validate_bundle_source(&base.advertisement, prepared, now_ms)?; if base.advertisement.bundle == Some(prepared.head) { + validate_selected_coverage(&base.advertisement, native_through)?; return Ok(base); } if base.advertisement.bundle != prepared.original { @@ -62,6 +90,11 @@ impl NodeDirectory { } let mut next = base.advertisement.clone(); next.bundle = Some(prepared.head); + if let (Some(log), Some(through)) = (&next.log, native_through) { + // Root materialization can cover later native ranges. Rebase + // without regressing either that frontier or heartbeat state. + next.log = Some(log.advance_tiered(next.node, through.max(log.tiered_through()))?); + } next.generation = next .generation .checked_add(1) @@ -71,6 +104,7 @@ impl NodeDirectory { Err(source) => match self.load(prepared.catalog.session, now_ms).await { Ok(Some(current)) if current.advertisement.bundle == Some(prepared.head) => { self.validate_bundle_source(¤t.advertisement, prepared, now_ms)?; + validate_selected_coverage(¤t.advertisement, native_through)?; return Ok(current); } Ok(Some(current)) @@ -106,6 +140,17 @@ impl NodeDirectory { } } +fn validate_selected_coverage(source: &NodeAdvertisement, through: Option) -> Result<()> { + if let (Some(log), Some(through)) = (&source.log, through) + && log.tiered_through() < through + { + return Err(Error::Node( + "selected bundle lacks canonical native coverage", + )); + } + Ok(()) +} + #[cfg(test)] pub(super) async fn load_catalog( layout: &cellule_ltx::CellStorageLayout, diff --git a/crates/cellule-runtime/src/node/bundle/tests/coverage.rs b/crates/cellule-runtime/src/node/bundle/tests/coverage.rs new file mode 100644 index 00000000..d0ff9133 --- /dev/null +++ b/crates/cellule-runtime/src/node/bundle/tests/coverage.rs @@ -0,0 +1,473 @@ +use super::*; +use crate::node::log_state::NodeLogStatus; + +pub(super) async fn enroll(f: &mut Fixture) { + let mut node = f.node.advertisement().clone(); + node.log = Some( + NodeLogStatus::open( + node.node, + EPOCH, + vec![NodeId::from_bytes([2; 16]), NodeId::from_bytes([3; 16])], + ) + .unwrap(), + ); + node.generation += 1; + f.node = f + .directory + .update_advertisement(&f.node, node, NOW) + .await + .unwrap(); +} + +struct NoExtraIo; +impl crate::node::durability::NodeLogAuthority for NoExtraIo { + fn activate<'a>(&'a self, _: u64) -> futures_util::future::BoxFuture<'a, Result<()>> { + panic!("bundle confirmation must not activate followers") + } + fn advance_coverage<'a>( + &'a self, + _: u64, + _: u64, + ) -> futures_util::future::BoxFuture<'a, Result<()>> { + panic!("bundle confirmation must not perform another authority CAS") + } + fn close<'a>( + &'a self, + _: &'a crate::node::log::NodeLogRetirementObservation, + ) -> futures_util::future::BoxFuture<'a, Result<()>> { + panic!("unused test closure") + } +} +impl crate::node::log_transport::NodeLogTransport for NoExtraIo { + fn append<'a>( + &'a self, + _: NodeId, + _: crate::node::log_transport::AppendRequest, + ) -> futures_util::future::BoxFuture<'a, Result> { + panic!("unused test append") + } + fn seal<'a>( + &'a self, + _: NodeId, + _: crate::node::log_transport::SealRequest, + ) -> futures_util::future::BoxFuture<'a, Result> { + panic!("unused test seal") + } + fn tail<'a>( + &'a self, + _: NodeId, + _: crate::node::log_transport::TailRequest, + ) -> futures_util::future::BoxFuture<'a, Result>> { + panic!("unused test tail") + } + fn retire<'a>( + &'a self, + _: NodeId, + _: crate::node::log_transport::RetireRequest, + ) -> futures_util::future::BoxFuture<'a, Result> { + panic!("unused test retirement") + } +} + +#[tokio::test] +async fn public_confirmation_and_later_root_proof_need_no_extra_native_cas() { + let mut f = Fixture::new().await; + enroll(&mut f).await; + let mut cell = f.cell(4).await; + let (_, frames, assignment) = f.append(&mut cell, 2); + let prepared = f + .directory + .prepare_node_bundle(&f.node, &frames, &[assignment], NOW) + .await + .unwrap(); + let (_, proofs) = f + .directory + .select_node_bundle(&f.node, &prepared, &f.lease, Limits::default(), NOW) + .await + .unwrap(); + let transport: Arc = Arc::new(NoExtraIo); + let shipper = crate::node::log_shipper::NodeLogShipper::new( + f.gate.clone(), + transport.clone(), + Limits::default(), + ) + .unwrap(); + let durability = crate::node::durability::NodeDurability::new( + f.gate.clone(), + shipper, + Arc::new(NoExtraIo), + transport, + f.lease.clone(), + ); + f.count.reset(); + assert_eq!(durability.confirm_bundle(&proofs).unwrap(), 1); + assert_eq!( + durability + .prove(assignment.ticket()) + .await + .unwrap() + .source(), + DurabilitySource::Bundle + ); + assert_eq!(f.count.put_requests(), 0); + let mut publisher = f.publisher(&cell); + let root = publisher.materialize_bundle(&proofs[0]).await.unwrap(); + assert_eq!(root.commit_sequence, 2); + let materialization_puts = f.count.put_requests(); + assert_eq!( + durability + .prove_object(assignment.ticket()) + .await + .unwrap() + .source(), + DurabilitySource::Object + ); + assert_eq!(f.count.put_requests(), materialization_puts); +} + +#[tokio::test] +async fn bundle_confirmation_preserves_the_source_of_root_only_gaps() { + let mut f = Fixture::new().await; + enroll(&mut f).await; + let mut a = f.cell(4).await; + let mut b = f.cell(5).await; + let mut c = f.cell(6).await; + let (_, mut frames, arange) = f.append(&mut a, 2); + let (_, bframes, brange) = f.append(&mut b, 2); + let (_, cframes, crange) = f.append(&mut c, 2); + frames.extend(bframes); + frames.extend(cframes); + let prepared = f + .directory + .prepare_node_bundle(&f.node, &frames, &[arange, brange, crange], NOW) + .await + .unwrap(); + let (_, proofs) = f + .directory + .select_node_bundle(&f.node, &prepared, &f.lease, Limits::default(), NOW) + .await + .unwrap(); + let (root, bundles): (Vec<_>, Vec<_>) = proofs + .into_iter() + .partition(|proof| proof.binding() == a.control.value().bundle_binding.unwrap()); + f.publisher(&a).materialize_bundle(&root[0]).await.unwrap(); + f.gate.prove_object(arange.ticket()).unwrap(); + assert_eq!( + confirm_selected_coverage(&f.gate, &f.lease, &bundles).unwrap(), + 3 + ); + assert_eq!( + f.gate.prove(arange.ticket()).await.unwrap().source(), + DurabilitySource::Object + ); + for assignment in [brange, crange] { + assert_eq!( + f.gate.prove(assignment.ticket()).await.unwrap().source(), + DurabilitySource::Bundle + ); + } + assert_eq!( + confirm_selected_coverage(&f.gate, &f.lease, &root).unwrap(), + 3 + ); + assert_eq!( + f.gate.prove(arange.ticket()).await.unwrap().source(), + DurabilitySource::Bundle + ); + assert_eq!( + f.gate + .confirmed_object_proof(arange.ticket()) + .unwrap() + .source(), + DurabilitySource::Object + ); +} + +#[tokio::test] +async fn selection_advances_native_coverage_with_one_cas_while_roots_lag() { + let mut f = Fixture::new().await; + enroll(&mut f).await; + let mut a = f.cell(4).await; + let mut b = f.cell(5).await; + let (_, mut frames, arange) = f.append(&mut a, 2); + let (_, bframes, brange) = f.append(&mut b, 2); + frames.extend(bframes); + f.count.reset(); + let prepared = f + .directory + .prepare_node_bundle(&f.node, &frames, &[arange, brange], NOW) + .await + .unwrap(); + assert_eq!(f.gate.tiered_through(), 0, "upload grants no coverage"); + let (selected, proofs) = f + .directory + .select_node_bundle(&f.node, &prepared, &f.lease, Limits::default(), NOW) + .await + .unwrap(); + assert_eq!(selected.advertisement().log().unwrap().tiered_through(), 2); + assert_eq!( + selected + .advertisement() + .bundle_head() + .unwrap() + .selected_through, + 2 + ); + assert_eq!( + f.count.put_requests(), + 2, + "one immutable upload and one combined authority CAS" + ); + assert_eq!( + f.gate.tiered_through(), + 0, + "selection is not local confirmation" + ); + assert_eq!( + confirm_selected_coverage(&f.gate, &f.lease, &proofs).unwrap(), + 2 + ); + for assignment in [arange, brange] { + let proof = f.gate.prove(assignment.ticket()).await.unwrap(); + assert_eq!(proof.ticket(), assignment.ticket()); + assert_eq!(proof.source(), DurabilitySource::Bundle); + } + assert_eq!( + confirm_selected_coverage(&f.gate, &f.lease, &proofs).unwrap(), + 2 + ); + assert_eq!( + f.count.put_requests(), + 2, + "local retry performs no second CAS" + ); + for cell in [&a, &b] { + assert_eq!( + cell.authority + .load(cell.control.value().cell) + .await + .unwrap() + .unwrap() + .value(), + cell.control.value() + ); + } +} + +#[tokio::test] +async fn cold_proofs_and_replacement_gates_cannot_authorize_local_confirmation() { + let mut f = Fixture::new().await; + enroll(&mut f).await; + let mut cell = f.cell(4).await; + let (_, frames, assignment) = f.append(&mut cell, 2); + let prepared = f + .directory + .prepare_node_bundle(&f.node, &frames, &[assignment], NOW) + .await + .unwrap(); + let (_, proofs) = f + .directory + .select_node_bundle(&f.node, &prepared, &f.lease, Limits::default(), NOW) + .await + .unwrap(); + let cold = f + .directory + .load_bundle_coverage(&cell.authority, &cell.control, Limits::default()) + .await + .unwrap(); + assert!(matches!( + confirm_selected_coverage(&f.gate, &f.lease, &[cold]), + Err(Error::Node("bundle proof grants reconstruction only")) + )); + let replacement = DurabilityGate::new( + SessionId::from_bytes([1; 16]), + NodeId::from_bytes([1; 16]), + EPOCH, + [NodeId::from_bytes([2; 16]), NodeId::from_bytes([3; 16])], + ) + .unwrap(); + let reserved = replacement.preview(frames.len() as u64).unwrap(); + let duplicate = replacement + .commit_frames(reserved, &frames) + .unwrap() + .unwrap(); + assert_eq!(duplicate.ticket(), assignment.ticket()); + assert!(matches!( + confirm_selected_coverage(&replacement, &f.lease, &proofs), + Err(Error::Node( + "selected capture belongs to another durability gate" + )) + )); + assert_eq!(replacement.tiered_through(), 0); + let replacement_lease = NodeLeaseGuard::new(NOW, NOW + 30_000).unwrap(); + assert!(matches!( + confirm_selected_coverage(&f.gate, &replacement_lease, &proofs), + Err(Error::Fenced) + )); + f.lease.fence(); + assert!(matches!( + confirm_selected_coverage(&f.gate, &f.lease, &proofs), + Err(Error::Fenced) + )); + assert_eq!(f.gate.tiered_through(), 0); +} + +#[tokio::test] +async fn one_foreign_assignment_rejects_the_complete_confirmation_batch() { + let mut f = Fixture::new().await; + enroll(&mut f).await; + let mut a = f.cell(4).await; + let mut b = f.cell(5).await; + let (_, mut frames, valid) = f.append(&mut a, 2); + let (_, bframes, _) = f.append(&mut b, 2); + let foreign = DurabilityGate::new( + SessionId::from_bytes([1; 16]), + NodeId::from_bytes([1; 16]), + EPOCH, + [NodeId::from_bytes([2; 16]), NodeId::from_bytes([3; 16])], + ) + .unwrap(); + foreign + .commit_frames(foreign.preview(frames.len() as u64).unwrap(), &frames) + .unwrap(); + let invalid = foreign + .commit_frames(foreign.preview(bframes.len() as u64).unwrap(), &bframes) + .unwrap() + .unwrap(); + frames.extend(bframes); + let prepared = f + .directory + .prepare_node_bundle(&f.node, &frames, &[valid, invalid], NOW) + .await + .unwrap(); + let (_, proofs) = f + .directory + .select_node_bundle(&f.node, &prepared, &f.lease, Limits::default(), NOW) + .await + .unwrap(); + assert!(matches!( + confirm_selected_coverage(&f.gate, &f.lease, &proofs), + Err(Error::Node( + "selected capture belongs to another durability gate" + )) + )); + assert_eq!( + f.gate.tiered_through(), + 0, + "valid sibling must not be confirmed" + ); +} + +#[tokio::test] +async fn local_confirmation_never_invents_coverage_for_a_skipped_selected_bundle() { + let mut f = Fixture::new().await; + enroll(&mut f).await; + let mut cell = f.cell(4).await; + let (_, first, assignment) = f.append(&mut cell, 2); + let proposal = f + .directory + .prepare_node_bundle(&f.node, &first, &[assignment], NOW) + .await + .unwrap(); + let (selected, first_proofs) = f + .directory + .select_node_bundle(&f.node, &proposal, &f.lease, Limits::default(), NOW) + .await + .unwrap(); + f.node = selected; + let (_, second, assignment) = f.append(&mut cell, 3); + let proposal = f + .directory + .prepare_node_bundle(&f.node, &second, &[assignment], NOW) + .await + .unwrap(); + let (_, second_proofs) = f + .directory + .select_node_bundle(&f.node, &proposal, &f.lease, Limits::default(), NOW) + .await + .unwrap(); + assert_eq!( + confirm_selected_coverage(&f.gate, &f.lease, &second_proofs).unwrap(), + 0 + ); + assert_eq!( + f.gate.prove(assignment.ticket()).await.unwrap().source(), + DurabilitySource::Bundle + ); + assert_eq!( + confirm_selected_coverage(&f.gate, &f.lease, &first_proofs).unwrap(), + 2 + ); +} + +#[tokio::test] +async fn shared_selection_rebases_without_regressing_later_root_coverage() { + let mut f = Fixture::new().await; + enroll(&mut f).await; + let mut cell = f.cell(4).await; + let (_, frames, assignment) = f.append(&mut cell, 2); + let proposal = f + .directory + .prepare_node_bundle(&f.node, &frames, &[assignment], NOW) + .await + .unwrap(); + let stale = f.node.clone(); + let later = f + .directory + .advance_log_coverage(&f.node, 2, NOW) + .await + .unwrap(); + f.node = later; + let now = f.heartbeat().await; + let (selected, _) = f + .directory + .select_node_bundle(&stale, &proposal, &f.lease, Limits::default(), now) + .await + .unwrap(); + assert_eq!(selected.advertisement().log().unwrap().tiered_through(), 2); + assert_eq!(selected.advertisement().issued_at_ms(), now); + assert_eq!( + selected.advertisement().expires_at_ms(), + f.node.advertisement().expires_at_ms() + ); + assert_eq!( + selected.advertisement().progress(), + f.node.advertisement().progress() + ); + assert_eq!( + selected + .advertisement() + .bundle_head() + .unwrap() + .selected_through(), + 1 + ); +} + +#[tokio::test] +async fn reconciliation_refuses_a_matching_head_without_its_native_frontier() { + let mut f = Fixture::new().await; + enroll(&mut f).await; + let mut cell = f.cell(4).await; + let (_, frames, assignment) = f.append(&mut cell, 2); + let proposal = f + .directory + .prepare_node_bundle(&f.node, &frames, &[assignment], NOW) + .await + .unwrap(); + // Historical versions could select a bundle without advancing native log + // coverage. Such a record remains recoverable, but is not a live CAS reply. + f.directory + .select_catalog(&f.node, &proposal, NOW) + .await + .unwrap(); + assert!(matches!( + f.directory + .select_node_bundle(&f.node, &proposal, &f.lease, Limits::default(), NOW) + .await, + Err(Error::Node( + "selected bundle lacks canonical native coverage" + )) + )); + assert_eq!(f.gate.tiered_through(), 0); +} diff --git a/crates/cellule-runtime/src/node/bundle/tests/faults.rs b/crates/cellule-runtime/src/node/bundle/tests/faults.rs index c5066546..66b2d2c2 100644 --- a/crates/cellule-runtime/src/node/bundle/tests/faults.rs +++ b/crates/cellule-runtime/src/node/bundle/tests/faults.rs @@ -14,6 +14,8 @@ pub(super) struct ReplyFault { pub(super) coverage_puts: AtomicUsize, pub(super) pin_started: tokio::sync::Notify, pub(super) pin_resume: tokio::sync::Notify, + pub(super) node_started: tokio::sync::Notify, + pub(super) node_resume: tokio::sync::Notify, } impl std::fmt::Display for ReplyFault { fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { @@ -59,6 +61,16 @@ impl ObjectStore for ReplyFault { if node_update { self.node_updates.fetch_add(1, Ordering::SeqCst); } + if node_update + && mode == 10 + && self + .mode + .compare_exchange(10, 0, Ordering::SeqCst, Ordering::SeqCst) + .is_ok() + { + self.node_started.notify_one(); + self.node_resume.notified().await; + } if path.as_ref().ends_with(".cnb") { self.coverage_puts.fetch_add(1, Ordering::SeqCst); } @@ -321,6 +333,7 @@ async fn lost_immutable_reply_grants_no_proof_and_retry_reuses_exact_bytes() { async fn lost_node_cas_reply_reconciles_only_the_exact_selected_head() { let faults = Arc::new(ReplyFault::default()); let mut f = Fixture::with_store(faults.clone()).await; + super::coverage::enroll(&mut f).await; let mut cell = f.cell(4).await; let (_, frames, assigned) = f.append(&mut cell, 2); let prepared = f @@ -336,4 +349,130 @@ async fn lost_node_cas_reply_reconciles_only_the_exact_selected_head() { .unwrap(); assert_eq!(selected.advertisement().bundle_head(), Some(prepared.head)); assert_eq!(proofs[0].commit_sequence(), 2); + assert_eq!(selected.advertisement().log().unwrap().tiered_through(), 1); + let updates = faults.node_updates.load(Ordering::SeqCst); + assert_eq!( + confirm_selected_coverage(&f.gate, &f.lease, &proofs).unwrap(), + 1 + ); + assert_eq!(faults.node_updates.load(Ordering::SeqCst), updates); +} + +#[tokio::test] +async fn lease_loss_during_combined_cas_cannot_release_original_captures() { + let faults = Arc::new(ReplyFault::default()); + let mut f = Fixture::with_store(faults.clone()).await; + super::coverage::enroll(&mut f).await; + let mut cell = f.cell(4).await; + let (_, frames, assigned) = f.append(&mut cell, 2); + let prepared = f + .directory + .prepare_node_bundle(&f.node, &frames, &[assigned], NOW) + .await + .unwrap(); + let directory = f.directory.clone(); + let node = f.node.clone(); + let lease = f.lease.clone(); + faults.mode.store(10, Ordering::SeqCst); + let selecting = tokio::spawn(async move { + directory + .select_node_bundle(&node, &prepared, &lease, Limits::default(), NOW) + .await + }); + faults.node_started.notified().await; + f.lease.fence(); + faults.node_resume.notify_one(); + assert!(matches!(selecting.await.unwrap(), Err(Error::Fenced))); + let selected = f + .directory + .load(SessionId::from_bytes([1; 16]), NOW) + .await + .unwrap() + .unwrap(); + assert_eq!(selected.advertisement().log().unwrap().tiered_through(), 1); + assert_eq!( + selected + .advertisement() + .bundle_head() + .unwrap() + .selected_through(), + 1 + ); + assert_eq!(f.gate.tiered_through(), 0); + let cold = f + .directory + .load_bundle_coverage(&cell.authority, &cell.control, Limits::default()) + .await + .unwrap(); + assert_eq!(cold.commit_sequence(), 2); + assert!(matches!( + confirm_selected_coverage(&f.gate, &f.lease, &[cold]), + Err(Error::Fenced) + )); + assert_eq!( + cell.authority + .load(cell.control.value().cell) + .await + .unwrap() + .unwrap() + .value() + .ltx_root() + .unwrap() + .commit_sequence, + 1 + ); +} + +#[tokio::test] +async fn cancelled_combined_cas_keeps_assignments_for_exact_retry() { + let faults = Arc::new(ReplyFault::default()); + let mut f = Fixture::with_store(faults.clone()).await; + super::coverage::enroll(&mut f).await; + let mut cell = f.cell(4).await; + let (_, frames, assigned) = f.append(&mut cell, 2); + let prepared = Arc::new( + f.directory + .prepare_node_bundle(&f.node, &frames, &[assigned], NOW) + .await + .unwrap(), + ); + let proposal = Arc::clone(&prepared); + let directory = f.directory.clone(); + let node = f.node.clone(); + let lease = f.lease.clone(); + faults.mode.store(10, Ordering::SeqCst); + let selecting = tokio::spawn(async move { + directory + .select_node_bundle(&node, &proposal, &lease, Limits::default(), NOW) + .await + }); + faults.node_started.notified().await; + selecting.abort(); + assert!(selecting.await.err().unwrap().is_cancelled()); + let unchanged = f + .directory + .load(SessionId::from_bytes([1; 16]), NOW) + .await + .unwrap() + .unwrap(); + assert_eq!( + unchanged.advertisement().bundle_head(), + f.node.advertisement().bundle_head() + ); + assert_eq!(unchanged.advertisement().log().unwrap().tiered_through(), 0); + assert_eq!(f.gate.progress().unwrap().issued_through, 1); + assert_eq!(f.gate.tiered_through(), 0); + let (_, proofs) = f + .directory + .select_node_bundle(&f.node, &prepared, &f.lease, Limits::default(), NOW) + .await + .unwrap(); + assert_eq!( + confirm_selected_coverage(&f.gate, &f.lease, &proofs).unwrap(), + 1 + ); + assert_eq!( + f.gate.prove(assigned.ticket()).await.unwrap().source(), + DurabilitySource::Bundle + ); } diff --git a/crates/cellule-runtime/src/node/bundle/tests/index/compatibility.rs b/crates/cellule-runtime/src/node/bundle/tests/index/compatibility.rs index cfbd4b53..ce767783 100644 --- a/crates/cellule-runtime/src/node/bundle/tests/index/compatibility.rs +++ b/crates/cellule-runtime/src/node/bundle/tests/index/compatibility.rs @@ -33,6 +33,7 @@ async fn inline_indexed_suffix_migrates_to_detached_history_with_the_same_pin() }, body, catalog, + assignments: Vec::new(), }; f.layout .store() @@ -94,6 +95,7 @@ async fn legacy_selected_catalog_is_read_and_migrated_without_changing_the_cell_ }, body, catalog, + assignments: Vec::new(), }; f.layout .store() diff --git a/crates/cellule-runtime/src/node/bundle/tests/mod.rs b/crates/cellule-runtime/src/node/bundle/tests/mod.rs index 736c20c3..d86cf2e9 100644 --- a/crates/cellule-runtime/src/node/bundle/tests/mod.rs +++ b/crates/cellule-runtime/src/node/bundle/tests/mod.rs @@ -15,6 +15,7 @@ use std::sync::Arc; const NOW: i64 = 1_000_000; const EPOCH: u64 = 2; +mod coverage; mod faults; mod index; mod lifecycle; diff --git a/crates/cellule-runtime/src/node/bundle/tests/recovery/cohort.rs b/crates/cellule-runtime/src/node/bundle/tests/recovery/cohort.rs index 0864377a..05691ebd 100644 --- a/crates/cellule-runtime/src/node/bundle/tests/recovery/cohort.rs +++ b/crates/cellule-runtime/src/node/bundle/tests/recovery/cohort.rs @@ -52,11 +52,10 @@ async fn full_recovery_stages_65_real_cell_checkpoints_before_one_terminal_node_ .bundle_head() .unwrap() .selected_through(); - f.node = f - .directory - .advance_log_coverage(&f.node, through, NOW) - .await - .unwrap(); + assert_eq!( + f.node.advertisement().log().unwrap().tiered_through(), + through + ); let mut suffix = Vec::new(); let mut fleet = Vec::new(); for cell in &mut cells { diff --git a/crates/cellule-runtime/src/node/bundle/tests/recovery/mod.rs b/crates/cellule-runtime/src/node/bundle/tests/recovery/mod.rs index aafd312d..495fb8ea 100644 --- a/crates/cellule-runtime/src/node/bundle/tests/recovery/mod.rs +++ b/crates/cellule-runtime/src/node/bundle/tests/recovery/mod.rs @@ -218,11 +218,20 @@ async fn verify_recovery(scenario: Scenario) { .prepare_node_bundle(&f.node, &prefix, &[arange, brange], NOW) .await .unwrap(); - let (selected, _) = f - .directory - .select_node_bundle(&f.node, &proposal, &f.lease, Limits::default(), NOW) - .await - .unwrap(); + let selected = if matches!(scenario, Scenario::Overlap | Scenario::Conflict) { + // Retain recovery coverage for historical stored catalogs whose native + // frontier still lagged their selected bundle. New selection is atomic. + f.directory + .select_catalog(&f.node, &proposal, NOW) + .await + .unwrap() + } else { + f.directory + .select_node_bundle(&f.node, &proposal, &f.lease, Limits::default(), NOW) + .await + .unwrap() + .0 + }; let selected_through = selected .advertisement() .bundle_head() @@ -235,11 +244,11 @@ async fn verify_recovery(scenario: Scenario) { } else { selected_through }; - f.node = f - .directory - .advance_log_coverage(&selected, coverage, NOW) - .await - .unwrap(); + assert_eq!( + selected.advertisement().log().unwrap().tiered_through(), + coverage + ); + f.node = selected; let (suffix, fleet) = if matches!(scenario, Scenario::SelectedOnly) { (Vec::new(), None) } else { diff --git a/crates/cellule-runtime/src/node/durability/mod.rs b/crates/cellule-runtime/src/node/durability/mod.rs index fa1b09af..74ea0a7c 100644 --- a/crates/cellule-runtime/src/node/durability/mod.rs +++ b/crates/cellule-runtime/src/node/durability/mod.rs @@ -309,7 +309,7 @@ impl NodeDurability { /// only after the batch's authority CAS succeeds under the original node lease. pub async fn prove_object(&self, ticket: CommitTicket) -> Result { self.confirm_objects(&[ticket]).await?; - let proof = self.gate.prove(ticket).await?; + let proof = self.gate.confirmed_object_proof(ticket)?; self.node_lease.check()?; if proof.source() != DurabilitySource::Object { return Err(Error::Node("object proof lost its durability race")); @@ -336,6 +336,19 @@ impl NodeDurability { Ok(()) } + /// Confirms exact original captures after verified shared bundle selection. + /// + /// This performs no storage I/O or authority CAS. Selection already persisted + /// the bundle and native-log frontier together. Cold reconstruction proofs, + /// foreign gates and replacement lease guards cannot authorize local ACKs. + /// The caller must still join actor/read/retry visibility before responding. + pub fn confirm_bundle( + &self, + proofs: &[crate::node::bundle::BundleCoverageProof], + ) -> Result { + crate::node::bundle::confirm_selected_coverage(&self.gate, &self.node_lease, proofs) + } + /// Returns this binding's exact enrolled log epoch. pub fn log_epoch(&self) -> Result { self.gate.log_epoch() diff --git a/crates/cellule-runtime/src/node/durability/object_coverage.rs b/crates/cellule-runtime/src/node/durability/object_coverage.rs index 65614eea..af9a46db 100644 --- a/crates/cellule-runtime/src/node/durability/object_coverage.rs +++ b/crates/cellule-runtime/src/node/durability/object_coverage.rs @@ -1,5 +1,5 @@ //! Coalesce completed roots while serializing authoritative coverage confirmation. -use std::collections::BTreeMap; +use std::collections::{BTreeMap, BTreeSet}; use super::{CommitTicket, DurabilityGate, Error, NodeLeaseGuard, NodeLogAuthority, Result}; @@ -44,10 +44,23 @@ impl ObjectCoverage { if tickets.is_empty() && self.pending()?.is_empty() { return Ok(()); } - let _flushing = tokio::select! { - guard = self.flushing.lock() => guard, - () = lease.wait_fenced() => return Err(Error::Fenced), + let covered = + gate.objects_are_covered(&self.pending()?.values().copied().collect::>())?; + let _flushing = if covered { + // A selected bundle may supersede staged root work. Join its old + // flusher before dropping that work, including after lease loss; + // this branch grants no new coverage or response proof. + self.flushing.lock().await + } else { + tokio::select! { + guard = self.flushing.lock() => guard, + () = lease.wait_fenced() => return Err(Error::Fenced), + } }; + self.discard_covered(gate)?; + if tickets.is_empty() && self.pending()?.is_empty() { + return Ok(()); + } lease.check()?; if !tickets.is_empty() && gate.objects_are_covered(tickets)? { return Ok(()); @@ -78,4 +91,18 @@ impl ObjectCoverage { .lock() .map_err(|_| Error::Node("node-log object coverage queue poisoned")) } + + fn discard_covered(&self, gate: &DurabilityGate) -> Result<()> { + let mut pending = self.pending()?; + if pending.is_empty() { + return Ok(()); + } + let uncovered = gate + .uncovered_objects(&pending.values().copied().collect::>())? + .into_iter() + .map(|ticket| ticket.first_sequence()) + .collect::>(); + pending.retain(|first, _| uncovered.contains(first)); + Ok(()) + } } diff --git a/crates/cellule-runtime/src/node/durability/tests.rs b/crates/cellule-runtime/src/node/durability/tests.rs index 9fc02617..909ab5d5 100644 --- a/crates/cellule-runtime/src/node/durability/tests.rs +++ b/crates/cellule-runtime/src/node/durability/tests.rs @@ -339,6 +339,43 @@ async fn object_proof_advances_authoritative_contiguous_coverage() { assert_eq!(authority.0.lock().unwrap().coverage, vec![(2, 1)]); } +#[tokio::test] +async fn selected_bundle_retires_already_covered_staged_root_work_before_lease_loss_drain() { + let gate = DurabilityGate::new(session(1), node(1), 2, [node(2)]).unwrap(); + let transport: Arc = Arc::new(ImmediateTransport); + let shipper = NodeLogShipper::new( + gate.clone(), + transport.clone(), + cellule_ltx::Limits::default(), + ) + .unwrap(); + let authority = Arc::new(RecordingAuthority::default()); + let lease = lease(); + let durability = NodeDurability::new( + gate.clone(), + shipper, + authority.clone(), + transport, + lease.clone(), + ); + let (_directory, cuts) = capture(); + let (ticket, assignment) = durability.submit_assigned(submission(&cuts)).await.unwrap(); + assert!(durability.object_coverage.stage(&gate, &[ticket]).unwrap()); + gate.confirm_bundle_ranges(&[assignment]).unwrap(); + lease.fence(); + durability + .object_coverage + .flush(&gate, authority.as_ref(), &lease, &[]) + .await + .unwrap(); + assert!(authority.0.lock().unwrap().coverage.is_empty()); + durability + .object_coverage + .flush(&gate, authority.as_ref(), &lease, &[]) + .await + .unwrap(); +} + #[tokio::test] async fn concurrent_object_proofs_persist_the_complete_contiguous_prefix() { for reverse in [false, true] { diff --git a/crates/cellule-runtime/src/node/lease.rs b/crates/cellule-runtime/src/node/lease.rs index cf929a74..02158c81 100644 --- a/crates/cellule-runtime/src/node/lease.rs +++ b/crates/cellule-runtime/src/node/lease.rs @@ -28,6 +28,10 @@ struct LeaseState { } impl NodeLeaseGuard { + pub(crate) fn same_lease(&self, other: &Self) -> bool { + Arc::ptr_eq(&self.inner, &other.inner) + } + /// Starts a watchdog from one successfully published lease observation. pub fn new(now_ms: i64, expires_at_ms: i64) -> Result { let remaining = lease_remaining(now_ms, expires_at_ms)?; diff --git a/crates/cellule-runtime/src/node/log/mod.rs b/crates/cellule-runtime/src/node/log/mod.rs index cf3c3e63..16f96fd0 100644 --- a/crates/cellule-runtime/src/node/log/mod.rs +++ b/crates/cellule-runtime/src/node/log/mod.rs @@ -1,6 +1,7 @@ //! Node log: durability gate, rotation barrier, and recovery overlays. use std::collections::{BTreeMap, BTreeSet, HashMap, HashSet}; use std::path::Path; +use std::sync::atomic::{AtomicU64, Ordering}; use std::sync::{Arc, Mutex}; use tokio::sync::Notify; @@ -21,6 +22,7 @@ pub use retirement::{ pub use recovery::*; pub(crate) const MAX_TICKET_FRAMES: u64 = 1_024; +static NEXT_GATE_INSTANCE: AtomicU64 = AtomicU64::new(1); /// Exact Cell writer identity carried by the ordered native frame lane. #[derive(Clone, Copy, Debug, PartialEq, Eq, Hash)] @@ -53,6 +55,7 @@ pub struct CellIssuedRange { /// command number. Its digest binds every assigned native frame in order. #[derive(Clone, Copy, Debug, PartialEq, Eq)] pub struct AssignedCommitRange { + gate_instance: u64, ticket: CommitTicket, scope: CellLogScope, first_commit: u64, @@ -61,6 +64,10 @@ pub struct AssignedCommitRange { digest: [u8; 32], } impl AssignedCommitRange { + pub(crate) const fn scope(&self) -> CellLogScope { + self.scope + } + /// Exact original native ticket. pub const fn ticket(&self) -> CommitTicket { self.ticket @@ -166,8 +173,10 @@ impl CommitTicket { pub enum DurabilitySource { /// The commit was covered by the enrolled node-log lane. Fleet, - /// The commit was covered by object storage. + /// The commit was covered by an exact materialized Cell root in object storage. Object, + /// A verified node bundle is selected in origin; the Cell root may lag. + Bundle, } /// Non-forgeable proof issued by the gate after one complete path wins. @@ -269,6 +278,7 @@ pub struct NodeLogProgress { } struct GateState { + instance: u64, leader_session: SessionId, leader_node: NodeId, log_epoch: u64, @@ -277,6 +287,9 @@ struct GateState { // Keep only completed sequences above the contiguous prefix. The prefix // proves older tickets without one allocation per historical frame. object_covered: BTreeSet, + // Merge adjacent exact confirmations. A dense selected prefix occupies + // one entry without retaining one source tag per historical native frame. + bundle_covered: BTreeMap, tiered_through: u64, next_sequence: u64, fleet_active: bool, @@ -329,14 +342,23 @@ impl DurabilityGate { return Err(Error::Node("invalid node-log ensemble")); } let follower_through = members.iter().map(|member| (*member, 0)).collect(); + // Not a persisted identity: exact assignments cannot be confirmed by a + // second in-process gate even when boot, epoch and ticket numbers match. + let instance = NEXT_GATE_INSTANCE + .fetch_update(Ordering::Relaxed, Ordering::Relaxed, |next| { + next.checked_add(1) + }) + .map_err(|_| Error::Node("durability gate instance overflow"))?; Ok(Self { inner: Arc::new(Mutex::new(GateState { + instance, leader_session, leader_node, log_epoch, members, follower_through, object_covered: BTreeSet::new(), + bundle_covered: BTreeMap::new(), tiered_through: 0, next_sequence: 1, fleet_active: false, @@ -445,6 +467,7 @@ impl DurabilityGate { match &mut assignment { None => { assignment = Some(AssignedCommitRange { + gate_instance: state.instance, ticket, scope: cell_scope, first_commit: frame.first_commit_sequence(), @@ -712,23 +735,54 @@ impl DurabilityGate { if state.fenced { return Err(Error::Fenced); } - for ticket in tickets { - for sequence in ticket.first_sequence..=ticket.last_sequence { - if sequence > state.tiered_through { - state.object_covered.insert(sequence); - } - } + mark_object_coverage(&mut state, tickets.iter().copied()); + let tiered_through = state.tiered_through; + drop(state); + self.changed.notify_waiters(); + Ok(tiered_through) + } + + pub(crate) fn confirm_bundle_ranges(&self, assignments: &[AssignedCommitRange]) -> Result { + let mut state = self.lock()?; + if state.fenced { + return Err(Error::Fenced); } - while let Some(next) = state.tiered_through.checked_add(1) { - if !state.object_covered.remove(&next) { - break; + for assignment in assignments { + validate_ticket(&state, assignment.ticket)?; + if assignment.gate_instance != state.instance { + return Err(Error::Node( + "selected capture belongs to another durability gate", + )); } - state.tiered_through = next; } - let tiered_through = state.tiered_through; + for assignment in assignments { + mark_bundle_coverage(&mut state, assignment.ticket); + } + mark_object_coverage( + &mut state, + assignments.iter().map(|assignment| assignment.ticket), + ); + let through = state.tiered_through; drop(state); self.changed.notify_waiters(); - Ok(tiered_through) + Ok(through) + } + + pub(crate) fn confirmed_object_proof(&self, ticket: CommitTicket) -> Result { + let state = self.lock()?; + validate_ticket(&state, ticket)?; + if state.fenced { + return Err(Error::Fenced); + } + if !object_covers(&state, ticket) { + return Err(Error::Node("object ticket remains unconfirmed")); + } + // This caller completed an exact Cell root CAS. A bundle that won the + // earlier generic race must not turn materialization into an error. + Ok(DurabilityProof { + ticket, + source: DurabilitySource::Object, + }) } pub(crate) fn preview_objects(&self, tickets: &[CommitTicket]) -> Result { @@ -754,7 +808,7 @@ impl DurabilityGate { Ok(tiered_through) } - /// Returns the highest sequence the follower lane has made durable. + /// Returns the contiguous authoritative object prefix, including selected bundles. #[must_use] pub fn tiered_through(&self) -> u64 { self.lock().map_or(0, |state| state.tiered_through) @@ -822,7 +876,16 @@ impl DurabilityGate { if object_covers(&state, ticket) { return Ok(Some(DurabilityProof { ticket, - source: DurabilitySource::Object, + source: if state + .bundle_covered + .range(..=ticket.first_sequence) + .next_back() + .is_some_and(|(_, through)| *through >= ticket.last_sequence) + { + DurabilitySource::Bundle + } else { + DurabilitySource::Object + }, })); } if state.fleet_active @@ -902,6 +965,42 @@ pub async fn close_node_log( directory.close_log(observed, &barrier, now_ms).await } +fn mark_bundle_coverage(state: &mut GateState, ticket: CommitTicket) { + let mut first = ticket.first_sequence; + let mut last = ticket.last_sequence; + if let Some((&before, &through)) = state.bundle_covered.range(..=first).next_back() + && through >= first.saturating_sub(1) + { + first = before; + last = last.max(through); + state.bundle_covered.remove(&before); + } + while let Some((&after, &through)) = state.bundle_covered.range(first..).next() { + if after > last.saturating_add(1) { + break; + } + last = last.max(through); + state.bundle_covered.remove(&after); + } + state.bundle_covered.insert(first, last); +} + +fn mark_object_coverage(state: &mut GateState, tickets: impl Iterator) { + for ticket in tickets { + for sequence in ticket.first_sequence..=ticket.last_sequence { + if sequence > state.tiered_through { + state.object_covered.insert(sequence); + } + } + } + while let Some(next) = state.tiered_through.checked_add(1) { + if !state.object_covered.remove(&next) { + break; + } + state.tiered_through = next; + } +} + fn object_covers(state: &GateState, ticket: CommitTicket) -> bool { ticket.last_sequence <= state.tiered_through || (ticket.first_sequence..=ticket.last_sequence).all(|sequence| { diff --git a/crates/cellule-runtime/src/publication/mod.rs b/crates/cellule-runtime/src/publication/mod.rs index daa9b8ac..55b5f827 100644 --- a/crates/cellule-runtime/src/publication/mod.rs +++ b/crates/cellule-runtime/src/publication/mod.rs @@ -1153,6 +1153,14 @@ pub(crate) struct PendingDurability { impl PendingDurability { pub(crate) async fn prove(&self) -> Result { let proof = self.durability.prove(self.ticket).await?; + if proof.source() == crate::node::log::DurabilitySource::Bundle { + // Ordinary actors still confirm visibility through per-Cell roots. + // Their existing object fallback must win until exact bundle proofs + // are connected to command, query, retry and capture release. + return Err(Error::Node( + "shared bundle actor response path is not installed", + )); + } if proof.source() == crate::node::log::DurabilitySource::Fleet { self.telemetry .durability_proof(proof.source(), self.submitted_at.elapsed()); diff --git a/crates/cellule-runtime/src/publication/tests.rs b/crates/cellule-runtime/src/publication/tests.rs index e06a4426..474a46b9 100644 --- a/crates/cellule-runtime/src/publication/tests.rs +++ b/crates/cellule-runtime/src/publication/tests.rs @@ -285,6 +285,58 @@ fn coverage_pending( }) } +#[tokio::test] +async fn ordinary_pending_response_refuses_bundle_until_visibility_is_integrated() { + use crate::node::log::DurabilitySource; + let (durability, gate, _) = coverage_binding(1); + let scratch = tempfile::tempdir().unwrap(); + let mut db = Db::open( + &scratch.path().join("bundle-source.sqlite"), + Limits::default(), + ) + .unwrap(); + db.transaction(|tx| tx.execute_batch("CREATE TABLE outcomes(v)")) + .unwrap(); + let cuts = db.capture().unwrap(); + let ticket = gate.preview(cuts.segments.len() as u64).unwrap(); + let frames = cuts + .segments + .iter() + .enumerate() + .map(|(offset, segment)| { + cellule_ltx::encode_node_frame( + cellule_ltx::NodeFrameScope { + leader_session: [1; 16], + log_epoch: 2, + node_sequence: ticket.first_sequence() + offset as u64, + application: [9; 16], + cell: [4; 32], + incarnation: [5; 16], + cell_epoch: 3, + commit_sequence: 1, + }, + segment.info().clone(), + Bytes::from(std::fs::read(segment.path()).unwrap()), + Limits::default(), + ) + .unwrap() + }) + .collect::>(); + let assignment = gate.commit_frames(ticket, &frames).unwrap().unwrap(); + gate.confirm_bundle_ranges(&[assignment]).unwrap(); + assert_eq!( + gate.prove(ticket).await.unwrap().source(), + DurabilitySource::Bundle + ); + let pending = coverage_pending(&durability, ticket).unwrap(); + assert!(matches!( + pending.prove().await, + Err(Error::Node( + "shared bundle actor response path is not installed" + )) + )); +} + #[tokio::test] async fn one_coalesced_root_confirms_all_covered_tickets_with_one_authority_update() { let (durability, gate, authority) = coverage_binding(1); diff --git a/docs/bundle-coverage-implementation.md b/docs/bundle-coverage-implementation.md index bb041aa2..d8325628 100644 --- a/docs/bundle-coverage-implementation.md +++ b/docs/bundle-coverage-implementation.md @@ -32,7 +32,8 @@ Uploading an immutable object supplies no coverage proof. | --- | --- | | `NodeDirectory::initialize_bundle_lane` / `bind_bundle_cell` | Establish a boot/epoch catalog; reserve a provisional inventory entry, pin the original Cell's base/code/schema/writer, then open it before issuing bundled commands | | `NodeLogShipper::submit_assigned` / `NodeDurability::submit_assigned` | Use the existing bounded native shipping lane and return an opaque witness for every frame in a complete capture | -| `prepare_node_bundle` / `select_node_bundle` | Reject missing, overlapping, unassigned or cross-binding ranges; verify origin extents before selecting; reconcile only the exact head under the original lease | +| `prepare_node_bundle` / `select_node_bundle` | Reject missing, overlapping, unassigned or cross-binding ranges; verify origin extents; select the bundle and enrolled native coverage in one node CAS; reconcile only the exact head and coverage under the original lease | +| `NodeDurability::confirm_bundle` | Confirm only the exact original assigned captures and lease locally, with no storage I/O or second authority CAS; reject cold reconstruction proofs and replacement gates or guards | | `BundleCoverageProof` | Retain authenticated immutable locators, logical endpoint, SQLite position and original base; retain no capture bodies | | `load_bundle_coverage` | Reopen a selected suffix from the authority-pinned canonical catalog, including a fenced boot; grant no writer or follower-suffix closure | | `CellPublisher::materialize_bundle` | Reconstruct the exact overlay and use normal root preparation, lineage and Cell CAS independently of selection; bounded small tails reuse canonical native coalescing and packing | @@ -42,6 +43,23 @@ Uploading an immutable object supplies no coverage proof. | Node withdrawal/maintenance | Refuse unresolved bindings; stale fencing preserves the catalog head; one bundle-bound boot cannot rotate its native log to another epoch | | Backup and collection | Backup refuses bound Cells. Coverage objects have no deletion path; this is retention, not a qualified collection implementation | +Live confirmation now reports `DurabilitySource::Bundle` separately from a +materialized Cell root. Adjacent exact ranges merge into compact source +intervals; root-only gaps keep their original source. A later exact root CAS +can still return an Object proof. Ordinary actors reject the Bundle source and +wait for their existing object fallback until command/read/retry visibility is +integrated. A boot without enrolled native log state supplies reconstruction +proofs only. Root work superseded by confirmed coverage joins its old flusher +before removing already-covered queue entries, including after lease loss. + +The two-Cell regression verifies one immutable upload plus one combined node +CAS while Cell roots lag; exact local confirmation and retry add zero PUTs. +Cases also cover lost replies, heartbeat rebasing with later root coverage, +cancellation, lease loss, foreign gates and cold proofs. Historical catalogs +whose coverage watermark lagged remain tested through cold recovery; a +matching head alone cannot reconcile a new live selection. This is protocol +and component I/O evidence, not an application TPS improvement. + Failed-owner recovery now joins dependency-verified selected prefixes and the sealed follower witness in the same file-backed builder. Complete shard inventory includes selected-only Cells; identical native overlap is skipped, @@ -151,6 +169,14 @@ obligations. See the [node performance design](../crates/cellule-runtime/docs/wr ## Verification and remaining delivery +The atomic-coverage snapshot passed all twelve contributor checks: **1,940 +top-level workspace cases passed, 38 ignored; 60 local LTX cases passed**. +The first full run timed out at an unchanged five-second follower-backlog drain. +That exact binary passed the case alone in 0.91 seconds; the identical frozen +source then passed the complete workspace rerun with the same four test workers +and deadlines. Failed runs, source hashes and count provenance remain outside +Git. No new application TPS was measured, and ordinary bundle ACKs stay disabled. + The detached-history regression first fails at the original 32-reference ceiling. With the new representation, one actual Cell among a **2,000-binding catalog** selects 215 commands without checkpoint and cold-restores its seed and From b6e8a7eaf96a12957caef07050b228ab149b13f6 Mon Sep 17 00:00:00 2001 From: forhappy Date: Wed, 7 Oct 2026 23:13:27 -0700 Subject: [PATCH 025/102] Report matched application TPS and latency at 4957985 --- docs/bundle-coverage-implementation.md | 9 +- docs/pr67-write-measurement-4957985.md | 119 +++++++++++++++++++++++++ 2 files changed, 125 insertions(+), 3 deletions(-) create mode 100644 docs/pr67-write-measurement-4957985.md diff --git a/docs/bundle-coverage-implementation.md b/docs/bundle-coverage-implementation.md index d8325628..d964ac75 100644 --- a/docs/bundle-coverage-implementation.md +++ b/docs/bundle-coverage-implementation.md @@ -175,7 +175,10 @@ The first full run timed out at an unchanged five-second follower-backlog drain. That exact binary passed the case alone in 0.91 seconds; the identical frozen source then passed the complete workspace rerun with the same four test workers and deadlines. Failed runs, source hashes and count provenance remain outside -Git. No new application TPS was measured, and ordinary bundle ACKs stay disabled. +Git. A subsequent [fresh application measurement](pr67-write-measurement-4957985.md) +at `4957985` observed 344.19 Fleet-configured TPS and 147.98 Bucket TPS, with +failed qualification. Fleet successful-write p99 regressed, and ordinary bundle +ACKs stay disabled; component I/O reductions do not establish application gains. The detached-history regression first fails at the original 32-reference ceiling. With the new representation, one actual Cell among a **2,000-binding @@ -205,8 +208,8 @@ The frozen detached-history and streaming-materialization source passed all twelve contributor checks: **1,908 workspace tests passed, 38 ignored; 60 local LTX tests passed**. Its 39 focused bundle/index tests include the 215-command exact-root and outcome regression. The file-backed recovery test also passes. -These results verify correctness and component work; the latest source has no -new end-to-end TPS measurement. Ordinary application responses still use +These results verify correctness and component work; this milestone did not +include an end-to-end TPS measurement. Ordinary application responses still use per-Cell publication, and production bundle ACK integration remains unfinished. The combined selected-prefix/follower recovery snapshot passed all twelve diff --git a/docs/pr67-write-measurement-4957985.md b/docs/pr67-write-measurement-4957985.md new file mode 100644 index 00000000..10e094c7 --- /dev/null +++ b/docs/pr67-write-measurement-4957985.md @@ -0,0 +1,119 @@ +# PR67 application write measurements at 4957985 + +**Performance parity is not delivered.** One fresh matched pair increased +Fleet-configured completed throughput by 19.8%, but successful-write p99 became +4.4 times worse, Fleet rotation/fallback occurred, and audit/drain failed. +Bucket throughput decreased 2.3%. These observations do not establish a +repeatable performance improvement. + +## Workload and provenance + +Measurements ran on 2026-10-08 UTC against Cellule +`49579857ac4e5b9015ecb88b0a1a78e283154460`, the prior measured WAL NORMAL source +`075b2cd45cb2762f9795aab643e0a6ad14e4ca8d`, and pinned celld v0.6.1 +`f2bf648663a610eefde71f3547ad61e9b896b1f0`. The current application was freshly +release-built. The baseline reused its original verified release binary; both +arms had identical client/auditor hashes, fixture bytes and pinned images. +The comparison verified the same Docker host and loaded runner. + +Each case used 1,000 uniformly selected Cells, 96-byte values, SQL INSERT plus +SELECT, a two-hour request/result ledger, 128 clients, a 128-offer queue, +30 seconds warmup and a 300-second window. This is a SQL ledger workload. +There was one repetition per point, with no read load. All serving roles, +provider and client shared one ARM64 Linux VM with 8 CPUs and nominal 16 GiB RAM. +Nodes retained the 4 GiB tmpfs ceiling; Cellule retained its 64 MiB memory and +1 GiB managed disk budgets. Another VM was already running on the physical host; +host isolation and A/A variance were not established. + +The original 2-CPU/2-GiB RustFS fixture was OOM-killed during the first current +Fleet window. Its observed 302.86 TPS is **invalid for a performance comparison**. +That failed run, its failed 110,112-ACK warm audit, and interrupted cleanup remain +in the evidence. The remaining original-profile cases were not run. + +The six cases below use a separate **provider-headroom diagnostic**: RustFS +kept its 2-CPU limit but received an 8-GiB memory ceiling. Every arm used that +same setting and fresh Linux Docker volumes. The external runner records this +distinct profile and forces `diagnostic=true`. Repository qualification +profiles, thresholds, durations and drain deadlines were not changed. Preflight +launch failures before traffic also remain retained. The larger-provider +diagnostic cannot qualify the original profile or physical-device durability. + +## Completed window results + +TPS counts successful logical writes completed inside the 300-second window. +Late responses, HTTP errors and dropped offers do not count as completed TPS. + +| Configured durability / offered writes per second | Prior Cellule TPS | Current Cellule TPS | celld TPS | +| --- | ---: | ---: | ---: | +| Fleet / 15,000 | 287.28 | 344.19 | 1,285.58 | +| Bucket / 2,000 | 151.40 | 147.98 | 553.78 | + +Current Cellule achieved 26.8% of celld's Fleet-configured throughput and 26.7% +of its Bucket throughput. **None passed delivery/latency qualification.** +These overloaded completion rates are not sustainable capacities. + +| System / durability | Successful-write scheduled p99 ms | All-attempt scheduled p99 ms | Window errors | Window queue drops | +| --- | ---: | ---: | ---: | ---: | +| Prior Cellule / Fleet | 595.18 | 165.3 | 3,316,987 | 1,096,829 | +| Current Cellule / Fleet | 2,625.04 | 159.0 | 2,752,505 | 1,643,987 | +| celld / Fleet | 372.62 | 217.6 | 1,477,993 | 2,636,334 | +| Prior Cellule / Bucket | 7,501.76 | 7,501.8 | 0 | 554,323 | +| Current Cellule / Bucket | 7,889.54 | 7,889.6 | 0 | 555,350 | +| celld / Bucket | 1,214.48 | 1,214.5 | 0 | 433,609 | + +Successful-write percentiles use exact nearest-rank journal values for successful +measured offers, including late responses. There were 252 late current Fleet +successes, zero prior/celld Fleet late successes, and 256 late successes in each +Bucket case. All-attempt histograms include errors, use 100-us resolution and +exclude dropped offers. Fast overload errors conceal successful-write latency. +Every case's journal counts reconcile with attempts, errors, window successes +and total offered requests. Warmup failures/drops remain in the external report. + +## Audits, mode stability and drain + +| Case | ACK read/retry verification | Shutdown / cold recovery | +| --- | --- | --- | +| Prior Cellule / Fleet | 96,373 checked; 465 HTTP 503 errors; 95,908 exact retries passed | Owner exceeded 120-second deadline; cold not reached | +| Current Cellule / Fleet | 119,690 checked; 3,273 HTTP 503 errors; 116,417 exact retries passed | Owner exceeded 120-second deadline; cold not reached | +| celld / Fleet | All 463,837 checks returned HTTP 500 after owner scratch exhaustion | Cold not reached | +| Prior Cellule / Bucket | All 50,569 ACKs passed both warm and cold read/exact retry | Drain 18.53 seconds; cold contract retry passed | +| Current Cellule / Bucket | All 50,362 ACKs passed both warm and cold read/exact retry | Drain 6.37 seconds; cold contract retry passed | +| celld / Bucket | All 182,209 ACKs passed both warm and cold read/exact retry | Drain 10.63 seconds; cold contract retry passed | + +Prior Cellule maintained active Fleet throughout the sampled window. Current +Cellule entered rotation and disabled Fleet shipping by the final sample, so +its configured-Fleet result includes fallback and cannot claim steady Fleet +capacity. Its native tiered frontier stopped at 60,025 while issuance reached +116,564. Live filesystem evidence records celld's owner tmpfs at 100%, while +each follower used about 16 MiB. Audit HTTP failures are availability failures; +they do not establish acknowledged data loss. + +## Architectural implications and remaining qualification + +Ordinary application bundle ACKs remain disabled. The measured Bucket fixture +has no active native node log; object-only native publication still needs +integration. Shared verified selection must enter command/read/retry visibility +and exact capture release, with admitted materializers and complete-range drain. + +Successful provider PUTs per completed write were 1.75 to 2.10 for prior/current +Fleet and 3.733 to 3.727 for prior/current Bucket. These window ratios include +publication and coordination, exclude trailing drain and SDK-internal retries, +and are affected by outstanding work. They do not prove the complete lifecycle +cost target. Cellule's publication debt grew materially during Fleet load; +rotation, overload availability and drain remain concrete failures. + +The next performance claim requires the real application path plus the existing +[qualification gates](write-performance-proposal.md): three matched five-minute +repetitions, zero errors/drops, stable frontiers/debt, all-ACK cold recovery, +complete drain and read guardrails. Component correctness and reduced helper +I/O are not substitutes for those measurements. + +Raw builds, manifests, clients, journals, exact percentile collector, failed +runs and retained provider volumes remain outside Git under +`cellule-write-perf-8ad1`. The external evidence index is +`tps-4957985-evidence-index.json`; normalized results are +`tps-4957985-headroom-results.json`. The standard `scripts/perf` build/run/report/ +compare workflow is documented in its [runbook](../scripts/perf/README.md). +The retained diagnostic runner differs only in provider memory, explicit +diagnostic provenance, and the relocated output-path guard; it is not a +qualification-profile change. From 67cde414d5467d1119f3c3ad2bf67d22b21cdaff Mon Sep 17 00:00:00 2001 From: forhappy Date: Wed, 7 Oct 2026 23:16:16 -0700 Subject: [PATCH 026/102] Distinguish sampled audit status codes from verified counts --- docs/pr67-write-measurement-4957985.md | 10 ++++++---- 1 file changed, 6 insertions(+), 4 deletions(-) diff --git a/docs/pr67-write-measurement-4957985.md b/docs/pr67-write-measurement-4957985.md index 10e094c7..5b9d4c34 100644 --- a/docs/pr67-write-measurement-4957985.md +++ b/docs/pr67-write-measurement-4957985.md @@ -73,9 +73,9 @@ and total offered requests. Warmup failures/drops remain in the external report. | Case | ACK read/retry verification | Shutdown / cold recovery | | --- | --- | --- | -| Prior Cellule / Fleet | 96,373 checked; 465 HTTP 503 errors; 95,908 exact retries passed | Owner exceeded 120-second deadline; cold not reached | -| Current Cellule / Fleet | 119,690 checked; 3,273 HTTP 503 errors; 116,417 exact retries passed | Owner exceeded 120-second deadline; cold not reached | -| celld / Fleet | All 463,837 checks returned HTTP 500 after owner scratch exhaustion | Cold not reached | +| Prior Cellule / Fleet | 96,373 checked; 465 errors, sampled HTTP 503; 95,908 exact retries passed | Owner exceeded 120-second deadline; cold not reached | +| Current Cellule / Fleet | 119,690 checked; 3,273 errors, sampled HTTP 503; 116,417 exact retries passed | Owner exceeded 120-second deadline; cold not reached | +| celld / Fleet | 463,837 checked; all failed, sampled HTTP 500 after owner scratch exhaustion | Cold not reached | | Prior Cellule / Bucket | All 50,569 ACKs passed both warm and cold read/exact retry | Drain 18.53 seconds; cold contract retry passed | | Current Cellule / Bucket | All 50,362 ACKs passed both warm and cold read/exact retry | Drain 6.37 seconds; cold contract retry passed | | celld / Bucket | All 182,209 ACKs passed both warm and cold read/exact retry | Drain 10.63 seconds; cold contract retry passed | @@ -83,7 +83,9 @@ and total offered requests. Warmup failures/drops remain in the external report. Prior Cellule maintained active Fleet throughout the sampled window. Current Cellule entered rotation and disabled Fleet shipping by the final sample, so its configured-Fleet result includes fallback and cannot claim steady Fleet -capacity. Its native tiered frontier stopped at 60,025 while issuance reached +capacity. The auditor retains aggregate counts and only four error examples; +the sampled status codes do not classify every failed ACK. Its native tiered +frontier stopped at 60,025 while issuance reached 116,564. Live filesystem evidence records celld's owner tmpfs at 100%, while each follower used about 16 MiB. Audit HTTP failures are availability failures; they do not establish acknowledged data loss. From b1856728984781cee5398f8d7185fb88fde23993 Mon Sep 17 00:00:00 2001 From: forhappy Date: Wed, 7 Oct 2026 23:33:51 -0700 Subject: [PATCH 027/102] Avoid node coverage CAS for sparse published roots --- .../docs/write-performance-design.md | 9 ++++ .../src/node/durability/mod.rs | 6 ++- .../src/node/durability/object_coverage.rs | 18 +++++--- .../src/node/durability/tests.rs | 43 ++++++++++++++++++- 4 files changed, 67 insertions(+), 9 deletions(-) diff --git a/crates/cellule-runtime/docs/write-performance-design.md b/crates/cellule-runtime/docs/write-performance-design.md index bf052aae..806539ea 100644 --- a/crates/cellule-runtime/docs/write-performance-design.md +++ b/crates/cellule-runtime/docs/write-performance-design.md @@ -123,3 +123,12 @@ Exact local confirmation performs no second CAS and reports a distinct Bundle proof. Original gate and lease identity remain required; cold proofs grant reconstruction only. Ordinary actors continue waiting for their root fallback until command/read/retry visibility and capture release consume that proof. + +The live root fallback now avoids a node authority mutation when an exact Cell +root covers only a sparse range beyond an unpublished native gap. That root +grants its own object proof under the original lease; it does not advance follower +reclamation or permit rotation. Closing the gap still persists the complete new +contiguous frontier before confirming it locally. Failed or cancelled advancing +CAS work remains staged for retry and joined drain. This removes redundant +coordination work from the existing application path; bundle ACK integration +and the node capacity qualification remain open. diff --git a/crates/cellule-runtime/src/node/durability/mod.rs b/crates/cellule-runtime/src/node/durability/mod.rs index 74ea0a7c..0041e3cc 100644 --- a/crates/cellule-runtime/src/node/durability/mod.rs +++ b/crates/cellule-runtime/src/node/durability/mod.rs @@ -305,8 +305,10 @@ impl NodeDurability { /// Records an already-published object root and persists its contiguous watermark. /// /// Callers must complete the exact Cell root CAS before invoking this method. - /// Concurrent completions share coverage updates; local proofs become visible - /// only after the batch's authority CAS succeeds under the original node lease. + /// Concurrent completions share coverage updates. A new contiguous frontier + /// becomes visible only after its authority CAS succeeds under the original + /// node lease. Sparse roots grant their own exact object proofs without a + /// node mutation; they cannot advance reclamation or close an unpublished gap. pub async fn prove_object(&self, ticket: CommitTicket) -> Result { self.confirm_objects(&[ticket]).await?; let proof = self.gate.confirmed_object_proof(ticket)?; diff --git a/crates/cellule-runtime/src/node/durability/object_coverage.rs b/crates/cellule-runtime/src/node/durability/object_coverage.rs index af9a46db..98fe7c0e 100644 --- a/crates/cellule-runtime/src/node/durability/object_coverage.rs +++ b/crates/cellule-runtime/src/node/durability/object_coverage.rs @@ -69,14 +69,20 @@ impl ObjectCoverage { let Some(first) = tickets.first() else { return Ok(()); }; - // Roots arriving during this CAS remain staged for the next flusher. - // A cancelled/failed CAS retains the original tickets for retry or drain; - // staging alone never wakes proof waiters or permits log retirement. let through = gate.preview_objects(&tickets)?; - tokio::select! { - result = authority.advance_coverage(first.log_epoch(), through) => result?, - () = lease.wait_fenced() => return Err(Error::Fenced), + if through > gate.tiered_through() { + // Persist every advancing reclamation frontier before confirming it + // locally. Roots arriving during I/O remain staged; a cancelled or + // failed CAS retains this exact batch for retry or joined drain. + tokio::select! { + result = authority.advance_coverage(first.log_epoch(), through) => result?, + () = lease.wait_fenced() => return Err(Error::Fenced), + } } + // Each caller already selected its exact Cell root. Sparse completions + // need no node mutation: they change no reclamation frontier and cannot + // close the gap. A concurrent bundle frontier was itself selected before + // its local confirmation, so it also needs no duplicate coverage CAS. lease.check()?; gate.prove_objects(&tickets)?; let mut pending = self.pending()?; diff --git a/crates/cellule-runtime/src/node/durability/tests.rs b/crates/cellule-runtime/src/node/durability/tests.rs index 909ab5d5..53903c02 100644 --- a/crates/cellule-runtime/src/node/durability/tests.rs +++ b/crates/cellule-runtime/src/node/durability/tests.rs @@ -521,7 +521,48 @@ async fn batching_out_of_order_roots_preserves_unpublished_gaps() { )); durability.prove_object(gap).await.unwrap(); assert_eq!(gate.tiered_through(), 6); - assert_eq!(authority.0.lock().unwrap().coverage.last(), Some(&(2, 6))); + // A completed root beyond the unpublished gap grants its own object proof, + // but cannot change the node's reclamation watermark. Persist only advances. + assert_eq!(authority.0.lock().unwrap().coverage, vec![(2, 2), (2, 6)]); +} + +#[tokio::test] +async fn sparse_root_proofs_do_not_wait_for_node_authority_or_close_the_gap() { + let gate = DurabilityGate::new(session(1), node(1), 2, [node(2)]).unwrap(); + let transport: Arc = Arc::new(ImmediateTransport); + let shipper = NodeLogShipper::new( + gate.clone(), + transport.clone(), + cellule_ltx::Limits::default(), + ) + .unwrap(); + let authority = Arc::new(RecordingAuthority(Mutex::new(AuthorityState { + reject_coverage: true, + ..AuthorityState::default() + }))); + let durability = + NodeDurability::new(gate.clone(), shipper, authority.clone(), transport, lease()); + let tickets = (0..64).map(|_| gate.issue(1).unwrap()).collect::>(); + for ticket in tickets.iter().skip(1).rev() { + let proof = durability.prove_object(*ticket).await.unwrap(); + assert_eq!(proof.ticket(), *ticket); + assert_eq!(proof.source(), DurabilitySource::Object); + } + assert!(authority.0.lock().unwrap().coverage.is_empty()); + assert_eq!(gate.tiered_through(), 0); + assert!(matches!( + gate.begin_rotation(), + Err(Error::PendingPublication) + )); + // Closing the hole still needs the original authority CAS. Its failure + // cannot authorize prefix reclamation, the gap's proof, or shutdown. + assert!(durability.prove_object(tickets[0]).await.is_err()); + assert_eq!(gate.tiered_through(), 0); + assert!(!gate.objects_are_covered(&tickets[..1]).unwrap()); + authority.0.lock().unwrap().reject_coverage = false; + durability.prove_object(tickets[0]).await.unwrap(); + assert_eq!(authority.0.lock().unwrap().coverage, vec![(2, 64)]); + assert_eq!(gate.begin_rotation().unwrap().covered_through(), 64); } #[tokio::test] From e900bfae21fba2d073316707b18f4a80d292d7d2 Mon Sep 17 00:00:00 2001 From: forhappy Date: Thu, 8 Oct 2026 00:39:16 -0700 Subject: [PATCH 028/102] Report matched sparse-root write measurements and failures --- .../docs/failover-and-followers.md | 11 +- docs/bundle-coverage-implementation.md | 7 + docs/pr67-sparse-root-coverage-measurement.md | 131 ++++++++++++++++++ 3 files changed, 145 insertions(+), 4 deletions(-) create mode 100644 docs/pr67-sparse-root-coverage-measurement.md diff --git a/crates/cellule-runtime/docs/failover-and-followers.md b/crates/cellule-runtime/docs/failover-and-followers.md index ae643133..b94b6ba9 100644 --- a/crates/cellule-runtime/docs/failover-and-followers.md +++ b/crates/cellule-runtime/docs/failover-and-followers.md @@ -385,10 +385,13 @@ race through stale local state. Completed roots from independent Cells queue their exact node-log tickets while an object-coverage CAS is in flight. The next publisher confirms the queued -group with one serialized authority update, without a batching timer. Staging -does not release a local proof or bridge an unpublished sequence gap. Failed or -cancelled updates retain the original tickets for retry, including shutdown; -local confirmation follows a successful CAS and a fresh node-lease check. +group with one serialized authority update when its contiguous watermark +advances, without a batching timer. A sparse group whose watermark is unchanged +needs no node authority I/O: its already-selected exact Cell roots grant their +own object proofs after a fresh original node-lease check. They cannot bridge an +unpublished gap or permit log retirement. Staging alone grants no proof. Failed +or cancelled advancing updates retain the original tickets for retry, including +shutdown; local frontier confirmation follows a successful CAS and lease check. One coalesced Cell root stages its entire covered ticket set before that flush. Tickets are grouped by the original durability binding; equal epoch numbers alone cannot combine different bindings. Per-command proof telemetry remains diff --git a/docs/bundle-coverage-implementation.md b/docs/bundle-coverage-implementation.md index d964ac75..9b4e45ca 100644 --- a/docs/bundle-coverage-implementation.md +++ b/docs/bundle-coverage-implementation.md @@ -9,6 +9,13 @@ measures `7fc0793`; it does not exercise bundle-based responses or establish write parity. The [WAL NORMAL comparison](pr67-normal-wal-reevaluation.md) separately records seven completed diagnostic cases and an interrupted matrix. +The [latest sparse-root coverage measurement](pr67-sparse-root-coverage-measurement.md) +exercises the ordinary application path at `b185672`: 378.88 completed writes/s +versus 282.43 before the change in one five-minute Fleet-configured pair. +Successful-write p99 improved, median latency worsened, and audit, drain and +debt gates failed. The paired celld owner exited during load. This is diagnostic +evidence, not repeatable improvement or qualified parity. + ## Celld reference and Cellule adaptation The reference is celld `f2bf648663a610eefde71f3547ad61e9b896b1f0`. diff --git a/docs/pr67-sparse-root-coverage-measurement.md b/docs/pr67-sparse-root-coverage-measurement.md new file mode 100644 index 00000000..ab360308 --- /dev/null +++ b/docs/pr67-sparse-root-coverage-measurement.md @@ -0,0 +1,131 @@ +# PR67 sparse-root coverage measurement + +**Performance parity is not delivered.** The live-path change observed 378.88 +successful writes/s versus 282.43 before it in one matched five-minute run. +Successful-write p99 improved, but median latency worsened, overload errors +increased, publication debt grew, and audit/drain failed. These rates are +overloaded completion observations, not sustainable capacities. + +## Change and safety evidence + +Production source `b1856728984781cee5398f8d7185fb88fde23993` avoids a node +authority GET/CAS when already-selected exact Cell roots prove sparse native +sequences without advancing the contiguous reclamation frontier. Each actual +frontier advance still persists authority coverage before local confirmation +and follower reclamation. Sparse proof cannot bridge an unpublished gap or +permit rotation; the original node lease must remain fresh. Failed advancing +updates retain their original staged tickets for retry and joined shutdown. + +The regression failed before the change: out-of-order roots called authority at +frontiers 0, 2 and 6 instead of only 2 and 6. A second test proves 63 sparse +roots without node I/O, rejects rotation across the missing first sequence, and +retains the gap-closing batch through authority failure before its successful +retry. All 16 adjacent durability tests pass. Ordinary actor bundle ACKs remain +disabled; this measures the existing per-Cell root path. + +## Matched workload + +Measurements ran on 2026-10-08 UTC. The baseline is the original verified +release binary of `49579857ac4e5b9015ecb88b0a1a78e283154460`; the candidate is a +fresh release build of `b185672`; celld v0.6.1 is pinned to +`f2bf648663a610eefde71f3547ad61e9b896b1f0`. Clients/auditors are byte-identical; +fixture, image, binary and source hashes are retained. The comparison verified +one Docker host and loaded runner. + +Each arm used 1,000 uniformly selected Cells, 96-byte values, SQL INSERT plus +SELECT with a two-hour durable request/result ledger, 128 clients, a 128-offer +queue, 30-second warmup and a 300-second window offering 15,000 writes/s. There +was **one repetition**, with no separate read load. Bucket, read-only and mixed +performance were not rerun for this change. + +Owner, two followers, client and provider shared one ARM64 Linux VM with 8 CPUs +and nominal 16 GiB RAM. Nodes retained 4 GiB tmpfs ceilings and Cellule's +64 MiB retained-memory/1 GiB managed-disk budgets. This is the separate +provider-headroom diagnostic: RustFS has 2 CPUs and an 8 GiB memory ceiling in +every arm. The prior original 2 GiB provider OOM failure remains in the +[earlier evidence](pr67-write-measurement-4957985.md). Qualification profiles and +deadlines remain unchanged. Another VM was running on the physical host; +serving-node isolation and physical-device durability are unqualified. + +## Actual completed windows + +TPS counts successful logical writes completed inside the 300-second window. +Percentiles below use exact nearest-rank journal values for successful measured +offers, including late responses. Fast error responses are excluded from these +percentiles; queue drops are excluded from both TPS and latency samples. + +| Metric | Baseline Cellule | Candidate Cellule | celld | +| --- | ---: | ---: | ---: | +| Completed writes/s | 282.43 | 378.88 | 1,258.30 | +| Successful writes inside window | 84,730 | 113,665 | 377,489 | +| Successful-write scheduled p50 ms | 89.40 | 111.77 | 85.32 | +| Successful-write scheduled p99 ms | 6,857.58 | 489.90 | 636.43 | +| Window errors | 1,913,111 | 3,240,146 | 1,085,585 | +| Queue drops | 2,501,905 | 1,146,156 | 3,036,926 | +| Late successful responses | 254 | 33 | 0 | + +Observed candidate throughput increased 34.1%; median latency worsened 25.0%. +Successful-write p99 still misses the 50 ms Fleet goal by about 9.8 times. +The unchanged baseline previously measured 344.19 TPS and now measures 282.43; +one pair cannot establish a repeatable causal improvement. celld's owner and +one follower exited with code 3 **during the window**, with Docker OOM flags +false. Its TPS is a failing-run observation, not a healthy capacity reference. +The exit cause is unresolved. Candidate TPS is 30.1% of that observed celld rate. + +## Audit, drain and publication failures + +| Case | Warm ACK read/exact-retry audit | Drain and cold recovery | +| --- | --- | --- | +| Baseline | 104,129 ACKs checked; 103,674 retries passed; 455 errors | Owner missed 120-second deadline; forced cleanup; cold not reached | +| Candidate | 126,856 ACKs checked; 126,327 retries passed; 529 errors | Owner missed 120-second deadline; forced cleanup; cold not reached | +| celld | 460,849 ACKs checked; no retries passed; 460,849 errors | Owner/follower exited during traffic; cold not reached | + +Cellule's four retained audit examples per case are HTTP 503; celld's are +transport failures. They do not classify every error or establish data loss. +The baseline rotated and disabled Fleet shipping during the window, so its +configured-Fleet result includes fallback. The candidate remained active in +all six samples, but its tiered frontier stalled: issued/follower-proven +131,956 versus tiered 61,296 at the end. This difference includes already-rooted +sparse completions; it is **not a count of missing or lost writes**. + +Candidate retained captures peaked at 69,107,248 bytes and ended at 40,185,871; +oldest publication age ended at 156,737 ms. The late-segment debt/age slopes +fail stability. Required post-audit provider filesystem and cold-lifecycle +snapshots are absent because audits failed; live samples do not substitute for +that evidence. All three cases fail qualification. + +## Measured I/O and next gap + +Node-authority-family window GET attempts fell from 5,795 to 975 and successful +PUTs from 5,885 to 1,065 (about 82% fewer PUTs). These totals include coordination +and all serving nodes. All-provider successful PUTs per completed command fell +from 2.088 to 1.920. Window ratios exclude trailing drain and SDK-internal +retries and are affected by outstanding work; they do not establish lifecycle +cost. Materialized commands per selected Cell root remain about 2.10, far from +the conditional 215-command checkpoint spacing. + +The next architectural work remains shared verified selection in actor +command/read/retry paths with exact capture release, admitted asynchronous root +materialization, complete issued-range drain/recovery and safe cross-Cell +collection. The [design and exit gates](../crates/cellule-runtime/docs/write-performance-design.md) +still require three matched repetitions, zero errors/drops, stable debt, +all-ACK cold recovery, complete drain and read guardrails before a parity claim. + +## Verification and retained evidence + +The frozen measured source passed all 11 contributor verification routes: +format, all-feature/all-target check, workspace tests, local LTX, warnings-denied +Clippy/API docs, boundaries/layout, document fences/links and SQL/peer contracts. +There were 1,941 top-level workspace cases including doctests, 38 ignored, plus +one nested subprocess success; 60 local LTX cases passed. Environment-dependent +ignored tests remain outside this result. + +Raw evidence stays outside Git under `cellule-write-perf-8ad1`. Normalized +results are `sparse-root-b185672-results.json`; the paired report is +`sparse-root-b185672-fleet15000-comparison.json`. Independent streaming +verification reconciled offer/attempt/error/success counts and exact successful +p99 against raw journals and verified 476 artifact hashes in +`sparse-root-b185672-evidence-index.json`. Build identities, failed-before logs, +isolated checks and retained Docker volumes remain available. The standard +[performance runbook](../scripts/perf/README.md) describes the build/run/report +workflow; the diagnostic runner and plan are retained with this evidence. From bb2f08986cb16ff0c0d477226767670f3fcd0a4f Mon Sep 17 00:00:00 2001 From: forhappy Date: Thu, 8 Oct 2026 01:23:05 -0700 Subject: [PATCH 029/102] Coalesce follower-proven roots before saturated preparation admission --- .../docs/failover-and-followers.md | 10 + .../docs/write-performance-design.md | 12 + .../src/cell/actor/admission.rs | 3 +- .../cellule-runtime/src/cell/actor/handle.rs | 11 + .../src/cell/actor/lifecycle/eviction.rs | 2 +- crates/cellule-runtime/src/cell/actor/mod.rs | 1 + .../cell/actor/publication_schedule/mod.rs | 83 ++++++ .../src/cell/actor/requests.rs | 24 +- crates/cellule-runtime/src/cell/actor/task.rs | 4 +- .../src/cell/actor/tasks/activation.rs | 2 +- .../cell/actor/tasks/maintenance_transfer.rs | 2 +- .../src/cell/actor/tasks/work.rs | 4 + .../cellule-runtime/src/cell/actor/tests.rs | 264 ++++++++++++++++++ crates/cellule-runtime/src/publication/mod.rs | 12 + .../src/publication/shared/mod.rs | 4 + 15 files changed, 431 insertions(+), 7 deletions(-) create mode 100644 crates/cellule-runtime/src/cell/actor/publication_schedule/mod.rs diff --git a/crates/cellule-runtime/docs/failover-and-followers.md b/crates/cellule-runtime/docs/failover-and-followers.md index b94b6ba9..1a37a5a8 100644 --- a/crates/cellule-runtime/docs/failover-and-followers.md +++ b/crates/cellule-runtime/docs/failover-and-followers.md @@ -397,6 +397,16 @@ Tickets are grouped by the original durability binding; equal epoch numbers alone cannot combine different bindings. Per-command proof telemetry remains scoped to every covered commit. +When the shared preparation lane is saturated, an already-active Fleet owner +can accumulate a root cohort outside that admission. Its oldest exact capture +must first have follower proof; a 50 ms follower stall starts normal fallback. +The cohort requests preparation at 32 physical captures, five seconds from its +oldest capture, or object fallback/backlog pressure. Drain, migration and fencing +wake the original waiting task. Capture and byte limits stay unchanged, and +deferment consumes the original retry grace. Owner reads and exact retries use +the canonical durable head; published-root readers and due hints await root CAS. +This remains per-Cell root publication; ordinary bundle ACKs are disabled. + **Retired lane collection.** Retired follower lanes keep their durable append-fence marker for ten minutes. The server then: diff --git a/crates/cellule-runtime/docs/write-performance-design.md b/crates/cellule-runtime/docs/write-performance-design.md index 806539ea..de87c806 100644 --- a/crates/cellule-runtime/docs/write-performance-design.md +++ b/crates/cellule-runtime/docs/write-performance-design.md @@ -132,3 +132,15 @@ contiguous frontier before confirming it locally. Failed or cancelled advancing CAS work remains staged for retry and joined drain. This removes redundant coordination work from the existing application path; bundle ACK integration and the node capacity qualification remain open. + +The existing Fleet root fallback now waits outside preparation admission when +the shared node lane is saturated. Only an already-active original Fleet lane +whose exact oldest capture obtains follower proof may defer. A cohort flushes +at 32 physical captures (half the unchanged per-Cell limit), five seconds from +its original oldest capture, object fallback, backlog pressure, drain, migration +or fencing. Slow follower proof starts ordinary root fallback after at most +50 ms. The original publication retry grace is not extended. Owner reads/retries +still use canonical durable-head confirmation; root readers and due hints follow +later root selection. This is bounded coalescing on the current root path, +not the completed node materializer or the 215-command bundle checkpoint model. +Application throughput and debt must be remeasured before claiming improvement. diff --git a/crates/cellule-runtime/src/cell/actor/admission.rs b/crates/cellule-runtime/src/cell/actor/admission.rs index a4e0fbcd..7f1850a5 100644 --- a/crates/cellule-runtime/src/cell/actor/admission.rs +++ b/crates/cellule-runtime/src/cell/actor/admission.rs @@ -96,7 +96,7 @@ pub(super) fn fence_active(active: &mut ActiveCell) { pub(super) fn fence_admission(admission: &CellAdmission) { admission.fenced.store(true, Ordering::Release); - admission.draining.store(true, Ordering::Release); + admission.begin_drain(); admission.requests.close(); admission.bytes.close(); } @@ -109,6 +109,7 @@ pub(super) fn new_cell_admission(owner_fence: crate::control::OwnerFence) -> Arc draining: AtomicBool::new(false), maintenance_quiescing: AtomicBool::new(false), fenced: AtomicBool::new(false), + publication_batch: super::publication_schedule::PublicationBatch::default(), }) } diff --git a/crates/cellule-runtime/src/cell/actor/handle.rs b/crates/cellule-runtime/src/cell/actor/handle.rs index 3498239a..9a38e15b 100644 --- a/crates/cellule-runtime/src/cell/actor/handle.rs +++ b/crates/cellule-runtime/src/cell/actor/handle.rs @@ -47,6 +47,14 @@ pub(super) struct CellAdmission { pub(super) draining: AtomicBool, pub(super) maintenance_quiescing: AtomicBool, pub(super) fenced: AtomicBool, + pub(super) publication_batch: super::publication_schedule::PublicationBatch, +} + +impl CellAdmission { + pub(super) fn begin_drain(&self) { + self.draining.store(true, Ordering::Release); + self.publication_batch.flush(); + } } pub(crate) struct CommandWork { @@ -465,6 +473,7 @@ impl CellHandle { { return Err(Error::CellDraining); } + self.admission.publication_batch.flush(); self.admission.requests.close(); self.admission.bytes.close(); let successor_admission = new_cell_admission(self.owner_fence()); @@ -507,6 +516,7 @@ impl CellHandle { { return Err(Error::CellDraining); } + self.admission.publication_batch.flush(); self.admission.requests.close(); self.admission.bytes.close(); let (reply, response) = oneshot::channel(); @@ -573,6 +583,7 @@ impl CellHandle { if retained >= limit.saturating_sub(limit / 4) || disk.used() >= disk.capacity().saturating_sub(disk.capacity() / 4) { + self.admission.publication_batch.flush(); return Err(Error::Capacity("publication backlog")); } } diff --git a/crates/cellule-runtime/src/cell/actor/lifecycle/eviction.rs b/crates/cellule-runtime/src/cell/actor/lifecycle/eviction.rs index 41aeb16a..bce92451 100644 --- a/crates/cellule-runtime/src/cell/actor/lifecycle/eviction.rs +++ b/crates/cellule-runtime/src/cell/actor/lifecycle/eviction.rs @@ -117,7 +117,7 @@ pub(in crate::cell::actor) fn begin_idle_cell_eviction( } return false; } - active.admission.draining.store(true, Ordering::Release); + active.admission.begin_drain(); active.admission.requests.close(); active.admission.bytes.close(); active.drain = reply.map(DrainReply::Unit); diff --git a/crates/cellule-runtime/src/cell/actor/mod.rs b/crates/cellule-runtime/src/cell/actor/mod.rs index ddfa53b3..08099f03 100644 --- a/crates/cellule-runtime/src/cell/actor/mod.rs +++ b/crates/cellule-runtime/src/cell/actor/mod.rs @@ -24,6 +24,7 @@ mod acquire; mod acquire_resume; mod acquisition_observer; mod prefix; +mod publication_schedule; mod serving; pub use acquisition_observer::{AcquisitionObservation, AcquisitionObserver}; pub use serving::CellServingObservation; diff --git a/crates/cellule-runtime/src/cell/actor/publication_schedule/mod.rs b/crates/cellule-runtime/src/cell/actor/publication_schedule/mod.rs new file mode 100644 index 00000000..4fe19aeb --- /dev/null +++ b/crates/cellule-runtime/src/cell/actor/publication_schedule/mod.rs @@ -0,0 +1,83 @@ +//! Fleet-proven root cohorts wait outside scarce preparation admission. + +use std::{ + sync::atomic::{AtomicBool, AtomicUsize, Ordering}, + time::{Duration, Instant}, +}; + +use tokio::sync::Notify; + +use super::{CellAdmission, MAX_PENDING_PUBLICATIONS, PendingDurability}; + +const BATCH_CAPTURES: usize = MAX_PENDING_PUBLICATIONS / 2; +const BATCH_AGE: Duration = Duration::from_secs(5); +const FOLLOWER_WAIT: Duration = Duration::from_millis(50); + +#[derive(Default)] +pub(super) struct PublicationBatch { + queued: AtomicUsize, + flush: AtomicBool, + changed: Notify, +} + +impl PublicationBatch { + pub(super) fn queued(&self, count: usize, object_fallback: bool) { + self.queued.store(count, Ordering::Release); + if object_fallback { + self.flush.store(true, Ordering::Release); + } + self.changed.notify_one(); + } + + pub(super) fn flush(&self) { + self.flush.store(true, Ordering::Release); + self.changed.notify_one(); + } + + pub(super) fn reset(&self, count: usize, object_fallback: bool) { + self.queued.store(count, Ordering::Release); + self.flush.store(object_fallback, Ordering::Release); + } + + fn ready(&self, admission: &CellAdmission) -> bool { + admission.draining.load(Ordering::Acquire) + || admission.fenced.load(Ordering::Acquire) + || self.flush.load(Ordering::Acquire) + || self.queued.load(Ordering::Acquire) >= BATCH_CAPTURES + } +} + +pub(super) async fn wait_for_batch( + admission: &CellAdmission, + durability: &PendingDurability, + submitted_at: Instant, +) { + // This is scheduling, never an ACK gate. Only an already-active original + // Fleet lane can defer a root, and this particular complete capture must + // first obtain its follower proof. Failure/stall starts ordinary fallback. + if !durability.fleet_active() + || admission.publication_batch.ready(admission) + || !matches!( + tokio::time::timeout(FOLLOWER_WAIT, durability.prove_fleet()).await, + Ok(Ok(())) + ) + { + return; + } + let deadline = submitted_at + BATCH_AGE; + loop { + let changed = admission.publication_batch.changed.notified(); + tokio::pin!(changed); + changed.as_mut().enable(); + if admission.publication_batch.ready(admission) + || !durability.fleet_active() + || Instant::now() >= deadline + { + return; + } + tokio::select! { + () = changed => {}, + () = tokio::time::sleep_until(deadline.into()) => return, + } + } +} diff --git a/crates/cellule-runtime/src/cell/actor/requests.rs b/crates/cellule-runtime/src/cell/actor/requests.rs index c73e9f74..10722c06 100644 --- a/crates/cellule-runtime/src/cell/actor/requests.rs +++ b/crates/cellule-runtime/src/cell/actor/requests.rs @@ -404,10 +404,25 @@ pub(super) fn start_publication( let generation = active.generation; let effect_id = active.begin_task(CoordinationEffect::Publication); let fleet_deadline = std::time::Instant::now() + FLEET_PUBLICATION_GRACE; + let oldest = active + .publications + .front() + .map(|queued| (queued.submitted_at, queued.durability.clone())); + let batch_admission = Arc::clone(&active.admission); tasks.spawn(async move { // Keep coverage in the bounded Cell queue while waiting. The publisher // token prevents another root or compaction from overtaking admission. - let result = publisher.admit_publication().await.map(|replica| { + let result = async { + if publisher.publication_saturated() + && let Some((submitted_at, Some(durability))) = oldest + { + publication_schedule::wait_for_batch(&batch_admission, &durability, submitted_at) + .await; + } + publisher.admit_publication().await + } + .await + .map(|replica| { Box::new(PublicationAdmission { replica, fleet_deadline, @@ -461,6 +476,13 @@ pub(super) fn start_admitted_publication( active.publisher = Some(*publisher); return; }; + active.admission.publication_batch.reset( + active.publications.len(), + active + .publications + .iter() + .any(|queued| queued.durability.is_none()), + ); let Some(newest) = coverage.last() else { active.finish_task(effect_id, CoordinationEffect::Publication); active.publisher = Some(*publisher); diff --git a/crates/cellule-runtime/src/cell/actor/task.rs b/crates/cellule-runtime/src/cell/actor/task.rs index 2fbffc8e..6c6069f9 100644 --- a/crates/cellule-runtime/src/cell/actor/task.rs +++ b/crates/cellule-runtime/src/cell/actor/task.rs @@ -309,7 +309,7 @@ pub(super) fn start_shutdown_drain( } active.inventory_refreshing = false; active.coordination.step(CoordinationInput::BeginShutdown); - active.admission.draining.store(true, Ordering::Release); + active.admission.begin_drain(); active.admission.requests.close(); active.admission.bytes.close(); if !active.queue.is_empty() { @@ -914,7 +914,7 @@ pub(super) fn handle_message( return; } } - active.admission.draining.store(true, Ordering::Release); + active.admission.begin_drain(); active.admission.requests.close(); active.admission.bytes.close(); active.admission = new_cell_admission(active.admission.owner_fence); diff --git a/crates/cellule-runtime/src/cell/actor/tasks/activation.rs b/crates/cellule-runtime/src/cell/actor/tasks/activation.rs index 4fe65acb..dcd88104 100644 --- a/crates/cellule-runtime/src/cell/actor/tasks/activation.rs +++ b/crates/cellule-runtime/src/cell/actor/tasks/activation.rs @@ -51,7 +51,7 @@ pub(super) fn handle_activated( return; } if shutdown.draining { - admission.draining.store(true, Ordering::Release); + admission.begin_drain(); admission.requests.close(); admission.bytes.close(); let _ = reply.send(Err(Error::RuntimeClosed)); diff --git a/crates/cellule-runtime/src/cell/actor/tasks/maintenance_transfer.rs b/crates/cellule-runtime/src/cell/actor/tasks/maintenance_transfer.rs index 586756a0..89388781 100644 --- a/crates/cellule-runtime/src/cell/actor/tasks/maintenance_transfer.rs +++ b/crates/cellule-runtime/src/cell/actor/tasks/maintenance_transfer.rs @@ -94,7 +94,7 @@ pub(super) fn handle( // work and refreshing readiness again. Keep semaphore owners // intact until confirm, so a definite refusal can restore // only native completion on this same exact activation. - active.admission.draining.store(true, Ordering::Release); + active.admission.begin_drain(); state.closing = true; } state.next_check = now + HYDRATION_TICK; diff --git a/crates/cellule-runtime/src/cell/actor/tasks/work.rs b/crates/cellule-runtime/src/cell/actor/tasks/work.rs index 994f49de..38862a44 100644 --- a/crates/cellule-runtime/src/cell/actor/tasks/work.rs +++ b/crates/cellule-runtime/src/cell/actor/tasks/work.rs @@ -98,6 +98,10 @@ pub(super) fn handle_executed( submitted_at: std::time::Instant::now(), proof, }); + active + .admission + .publication_batch + .queued(active.publications.len(), durability.is_none()); if durability.is_some() { active.unpublished_node_logs += 1; unpublished_node_log_bytes.fetch_add(retained_bytes, Ordering::AcqRel); diff --git a/crates/cellule-runtime/src/cell/actor/tests.rs b/crates/cellule-runtime/src/cell/actor/tests.rs index 502b0377..f20bf5fb 100644 --- a/crates/cellule-runtime/src/cell/actor/tests.rs +++ b/crates/cellule-runtime/src/cell/actor/tests.rs @@ -1,5 +1,269 @@ use super::{CellRuntimeStats, bounded_u32}; +#[tokio::test(flavor = "multi_thread")] +async fn pressured_fleet_roots_batch_without_hiding_reads_or_delaying_drain() { + use super::*; + use crate::cell::catalog::CellCatalog; + use crate::cell::executor::{HandlerOutcome, MutationIdentity}; + use crate::control::Owner; + use crate::identity::{IncarnationId, NamespaceId, NodeId, RequestId, TenantId}; + use crate::node::log::DurabilityGate; + use crate::node::log_shipper::NodeLogShipper; + use crate::node::log_transport::LocalFollowerTransport; + + let session = SessionId::from_bytes([87; 16]); + let incarnation = IncarnationId::from_bytes([88; 16]); + let member = NodeId::from_bytes([89; 16]); + let target = CellTarget::new( + TenantId::from_bytes([1; 16]), + ApplicationId::from_bytes([3; 16]), + NamespaceId::from_bytes([6; 16]), + b"pressured-fleet-roots", + ) + .unwrap(); + let layout = cellule_ltx::CellStorageLayout::new( + cellule_store::Store::new(Arc::new(object_store::memory::InMemory::new())), + object_store::path::Path::from("pressured-roots"), + *target.application().as_bytes(), + ); + let limits = cellule_ltx::Limits::default(); + let replica = cellule_ltx::CellReplica::new( + layout.clone(), + *target.cell_id().as_bytes(), + *incarnation.as_bytes(), + limits, + ) + .unwrap(); + let directory = tempfile::TempDir::new().unwrap(); + let follower = crate::FollowerStore::open( + directory.path().join("follower"), + limits, + cellule_ltx::DiskBudget::new(1 << 30), + ) + .unwrap(); + let transport = Arc::new(LocalFollowerTransport::new(member, follower)); + let gate = DurabilityGate::new(session, NodeId::from_bytes([90; 16]), 1, [member]).unwrap(); + let shipper = NodeLogShipper::new(gate.clone(), transport.clone(), limits).unwrap(); + let node_lease = NodeLeaseGuard::new(0, 60_000).unwrap(); + let runtime = CellRuntime::new_with_replica_host_requiring_node_lease( + SqlWorkerPool::new(1, 1).unwrap(), + 8 << 20, + session, + cellule_ltx::Host::default(), + ) + .unwrap(); + runtime.install_node_lease(node_lease.clone()).unwrap(); + runtime + .install_node_durability( + target.application(), + Arc::new(NodeDurability::new( + gate, + shipper, + Arc::new(BatchNodeAuthority), + transport, + node_lease, + )), + ) + .unwrap(); + let catalog = CellCatalog::new(layout.clone(), target.tenant()); + let code = Digest::from_bytes([91; 32]); + let proof = catalog + .provision(CatalogEntry::new(&target, CatalogRole::Application, code, 1).unwrap()) + .await + .unwrap(); + let authority = CellAuthority::new(layout); + let observed = authority + .create_initial( + &proof, + incarnation, + Owner { + session, + endpoint: "https://node.internal".into(), + }, + ) + .await + .unwrap(); + let handle = runtime + .bootstrap( + proof, + replica.clone(), + authority.clone(), + observed, + directory.path().join("cell.sqlite"), + |tx| { + tx.execute_batch( + "CREATE TABLE counter(value INTEGER); INSERT INTO counter VALUES (0)", + )?; + Ok(()) + }, + ) + .await + .unwrap(); + let identity = |id| MutationIdentity { + request_id: RequestId::from_bytes([id; 16]), + issued_at_ms: 10, + expires_at_ms: 60_000, + }; + let mut publications = runtime.subscribe_publications(); + let mut warmup_slots = Vec::new(); + for _ in 0..cellule_ltx::SHARED_PUBLICATION_ROWS { + warmup_slots.push(runtime.inner.shared_publication.admit().await.unwrap()); + } + let first = handle + .execute(identity(1), code, 20, 1_024, 1_024, |tx| { + tx.execute("UPDATE counter SET value = value + 1", [])?; + Ok(HandlerOutcome::Success(vec![1])) + }) + .await + .unwrap(); + assert_eq!(first.commit_sequence(), 1); + assert!( + runtime + .node_durability() + .unwrap() + .1 + .progress() + .unwrap() + .fleet_active + ); + drop(warmup_slots); + tokio::time::timeout(std::time::Duration::from_secs(5), publications.recv()) + .await + .unwrap() + .unwrap(); + tokio::time::timeout(std::time::Duration::from_secs(5), async { + loop { + let control = authority.load(target.cell_id()).await.unwrap().unwrap(); + if control.value().ltx_root().unwrap().commit_sequence == 1 { + break; + } + tokio::task::yield_now().await; + } + }) + .await + .unwrap(); + + // Model the bounded node lane's complete occupancy without retaining any + // synthetic captures. These are its real original preparation permits. + let mut occupied = Vec::new(); + for _ in 0..cellule_ltx::SHARED_PUBLICATION_ROWS { + occupied.push(runtime.inner.shared_publication.admit().await.unwrap()); + } + let mut acknowledged = Vec::new(); + for id in 2_u8..=12 { + let outcome = handle + .execute(identity(id), code, 20, 1_024, 1_024, move |tx| { + tx.execute("UPDATE counter SET value = value + 1", [])?; + Ok(HandlerOutcome::Success(vec![id])) + }) + .await + .unwrap(); + acknowledged.push(outcome); + } + drop(occupied); + // Once pressure clears, an already-proven cohort still owns its original + // bounded age. It does not immediately turn into another tiny root. + tokio::time::sleep(std::time::Duration::from_millis(200)).await; + let current = authority.load(target.cell_id()).await.unwrap().unwrap(); + assert_eq!(current.value().ltx_root().unwrap().commit_sequence, 1); + assert_eq!( + handle + .query(1_024, 1_024, |db| { + let value: i64 = db.query_row("SELECT value FROM counter", [], |row| row.get(0))?; + Ok(value.to_le_bytes().to_vec()) + }) + .await + .unwrap(), + 12_i64.to_le_bytes() + ); + for (id, expected) in (2_u8..=12).zip(acknowledged) { + assert_eq!( + handle + .execute(identity(id), code, 20, 1_024, 1_024, |_| { + panic!("exact retry must not execute SQL") + }) + .await + .unwrap(), + expected + ); + } + // Drain must wake the original delayed task rather than wait its five- + // second age. It still selects and releases the complete captured prefix. + tokio::time::timeout(std::time::Duration::from_secs(2), handle.drain()) + .await + .unwrap() + .unwrap(); + runtime.shutdown().await.unwrap(); + let root = authority + .load(target.cell_id()) + .await + .unwrap() + .unwrap() + .value() + .ltx_root() + .unwrap(); + assert_eq!(root.commit_sequence, 12); + let restored = directory.path().join("cold.sqlite"); + replica + .open_root(&root) + .await + .unwrap() + .restore(&restored) + .await + .unwrap(); + let db = cellule_ltx::rusqlite::Connection::open(restored).unwrap(); + assert_eq!( + db.query_row("SELECT value FROM counter", [], |row| row.get::<_, i64>(0)) + .unwrap(), + 12 + ); + assert_eq!( + db.query_row("SELECT count(*) FROM sys_requests", [], |row| row + .get::<_, i64>(0)) + .unwrap(), + 12 + ); + for id in 1_u8..=12 { + let actual = db + .query_row( + "SELECT operation_digest, result, commit_sequence FROM sys_requests WHERE request_id = ?1", + [RequestId::from_bytes([id; 16]).as_bytes().as_slice()], + |row| Ok((row.get::<_, Vec>(0)?, row.get::<_, Vec>(1)?, row.get::<_, i64>(2)?)), + ) + .unwrap(); + assert_eq!(actual, (code.as_bytes().to_vec(), vec![id], i64::from(id))); + } + let stats = runtime.stats(); + assert_eq!(stats.active_cells(), 0); + assert_eq!(stats.retained_bytes(), 0); +} + +struct BatchNodeAuthority; + +impl crate::node::durability::NodeLogAuthority for BatchNodeAuthority { + fn activate<'a>( + &'a self, + _epoch: u64, + ) -> futures_util::future::BoxFuture<'a, crate::Result<()>> { + Box::pin(async { Ok(()) }) + } + + fn advance_coverage<'a>( + &'a self, + _epoch: u64, + _through: u64, + ) -> futures_util::future::BoxFuture<'a, crate::Result<()>> { + Box::pin(async { Ok(()) }) + } + + fn close<'a>( + &'a self, + _retirement: &'a crate::node::log::NodeLogRetirementObservation, + ) -> futures_util::future::BoxFuture<'a, crate::Result<()>> { + Box::pin(async { Ok(()) }) + } +} + #[tokio::test] async fn empty_runtime_miss_does_not_wait_for_the_dispatcher() { use super::*; diff --git a/crates/cellule-runtime/src/publication/mod.rs b/crates/cellule-runtime/src/publication/mod.rs index 55b5f827..8c37df9d 100644 --- a/crates/cellule-runtime/src/publication/mod.rs +++ b/crates/cellule-runtime/src/publication/mod.rs @@ -161,6 +161,12 @@ impl CellPublisher { .await } + pub(crate) fn publication_saturated(&self) -> bool { + self.shared_publication + .as_ref() + .is_some_and(|publication| publication.saturated()) + } + pub(crate) fn durability_submitter(&self) -> CellDurabilitySubmitter { let control = self.observed.value(); CellDurabilitySubmitter { @@ -1151,6 +1157,12 @@ pub(crate) struct PendingDurability { } impl PendingDurability { + pub(crate) fn fleet_active(&self) -> bool { + self.durability + .progress() + .is_ok_and(|progress| progress.fleet_active) + } + pub(crate) async fn prove(&self) -> Result { let proof = self.durability.prove(self.ticket).await?; if proof.source() == crate::node::log::DurabilitySource::Bundle { diff --git a/crates/cellule-runtime/src/publication/shared/mod.rs b/crates/cellule-runtime/src/publication/shared/mod.rs index 84cc82e0..be8b3f81 100644 --- a/crates/cellule-runtime/src/publication/shared/mod.rs +++ b/crates/cellule-runtime/src/publication/shared/mod.rs @@ -56,6 +56,10 @@ pub(crate) struct SharedPublication { } impl SharedPublication { + pub(crate) fn saturated(&self) -> bool { + self.slots.available_permits() == 0 + } + pub(crate) fn new( resources: ResourceLedger, telemetry: crate::fleet::telemetry::CellTelemetryHandle, From ddb030c776e5d61a75b76d54e8610120be781df0 Mon Sep 17 00:00:00 2001 From: forhappy Date: Thu, 8 Oct 2026 01:59:26 -0700 Subject: [PATCH 030/102] Revert "Coalesce follower-proven roots before saturated preparation admission" This reverts commit bb2f08986cb16ff0c0d477226767670f3fcd0a4f. --- .../docs/failover-and-followers.md | 10 - .../docs/write-performance-design.md | 12 - .../src/cell/actor/admission.rs | 3 +- .../cellule-runtime/src/cell/actor/handle.rs | 11 - .../src/cell/actor/lifecycle/eviction.rs | 2 +- crates/cellule-runtime/src/cell/actor/mod.rs | 1 - .../cell/actor/publication_schedule/mod.rs | 83 ------ .../src/cell/actor/requests.rs | 24 +- crates/cellule-runtime/src/cell/actor/task.rs | 4 +- .../src/cell/actor/tasks/activation.rs | 2 +- .../cell/actor/tasks/maintenance_transfer.rs | 2 +- .../src/cell/actor/tasks/work.rs | 4 - .../cellule-runtime/src/cell/actor/tests.rs | 264 ------------------ crates/cellule-runtime/src/publication/mod.rs | 12 - .../src/publication/shared/mod.rs | 4 - 15 files changed, 7 insertions(+), 431 deletions(-) delete mode 100644 crates/cellule-runtime/src/cell/actor/publication_schedule/mod.rs diff --git a/crates/cellule-runtime/docs/failover-and-followers.md b/crates/cellule-runtime/docs/failover-and-followers.md index 1a37a5a8..b94b6ba9 100644 --- a/crates/cellule-runtime/docs/failover-and-followers.md +++ b/crates/cellule-runtime/docs/failover-and-followers.md @@ -397,16 +397,6 @@ Tickets are grouped by the original durability binding; equal epoch numbers alone cannot combine different bindings. Per-command proof telemetry remains scoped to every covered commit. -When the shared preparation lane is saturated, an already-active Fleet owner -can accumulate a root cohort outside that admission. Its oldest exact capture -must first have follower proof; a 50 ms follower stall starts normal fallback. -The cohort requests preparation at 32 physical captures, five seconds from its -oldest capture, or object fallback/backlog pressure. Drain, migration and fencing -wake the original waiting task. Capture and byte limits stay unchanged, and -deferment consumes the original retry grace. Owner reads and exact retries use -the canonical durable head; published-root readers and due hints await root CAS. -This remains per-Cell root publication; ordinary bundle ACKs are disabled. - **Retired lane collection.** Retired follower lanes keep their durable append-fence marker for ten minutes. The server then: diff --git a/crates/cellule-runtime/docs/write-performance-design.md b/crates/cellule-runtime/docs/write-performance-design.md index de87c806..806539ea 100644 --- a/crates/cellule-runtime/docs/write-performance-design.md +++ b/crates/cellule-runtime/docs/write-performance-design.md @@ -132,15 +132,3 @@ contiguous frontier before confirming it locally. Failed or cancelled advancing CAS work remains staged for retry and joined drain. This removes redundant coordination work from the existing application path; bundle ACK integration and the node capacity qualification remain open. - -The existing Fleet root fallback now waits outside preparation admission when -the shared node lane is saturated. Only an already-active original Fleet lane -whose exact oldest capture obtains follower proof may defer. A cohort flushes -at 32 physical captures (half the unchanged per-Cell limit), five seconds from -its original oldest capture, object fallback, backlog pressure, drain, migration -or fencing. Slow follower proof starts ordinary root fallback after at most -50 ms. The original publication retry grace is not extended. Owner reads/retries -still use canonical durable-head confirmation; root readers and due hints follow -later root selection. This is bounded coalescing on the current root path, -not the completed node materializer or the 215-command bundle checkpoint model. -Application throughput and debt must be remeasured before claiming improvement. diff --git a/crates/cellule-runtime/src/cell/actor/admission.rs b/crates/cellule-runtime/src/cell/actor/admission.rs index 7f1850a5..a4e0fbcd 100644 --- a/crates/cellule-runtime/src/cell/actor/admission.rs +++ b/crates/cellule-runtime/src/cell/actor/admission.rs @@ -96,7 +96,7 @@ pub(super) fn fence_active(active: &mut ActiveCell) { pub(super) fn fence_admission(admission: &CellAdmission) { admission.fenced.store(true, Ordering::Release); - admission.begin_drain(); + admission.draining.store(true, Ordering::Release); admission.requests.close(); admission.bytes.close(); } @@ -109,7 +109,6 @@ pub(super) fn new_cell_admission(owner_fence: crate::control::OwnerFence) -> Arc draining: AtomicBool::new(false), maintenance_quiescing: AtomicBool::new(false), fenced: AtomicBool::new(false), - publication_batch: super::publication_schedule::PublicationBatch::default(), }) } diff --git a/crates/cellule-runtime/src/cell/actor/handle.rs b/crates/cellule-runtime/src/cell/actor/handle.rs index 9a38e15b..3498239a 100644 --- a/crates/cellule-runtime/src/cell/actor/handle.rs +++ b/crates/cellule-runtime/src/cell/actor/handle.rs @@ -47,14 +47,6 @@ pub(super) struct CellAdmission { pub(super) draining: AtomicBool, pub(super) maintenance_quiescing: AtomicBool, pub(super) fenced: AtomicBool, - pub(super) publication_batch: super::publication_schedule::PublicationBatch, -} - -impl CellAdmission { - pub(super) fn begin_drain(&self) { - self.draining.store(true, Ordering::Release); - self.publication_batch.flush(); - } } pub(crate) struct CommandWork { @@ -473,7 +465,6 @@ impl CellHandle { { return Err(Error::CellDraining); } - self.admission.publication_batch.flush(); self.admission.requests.close(); self.admission.bytes.close(); let successor_admission = new_cell_admission(self.owner_fence()); @@ -516,7 +507,6 @@ impl CellHandle { { return Err(Error::CellDraining); } - self.admission.publication_batch.flush(); self.admission.requests.close(); self.admission.bytes.close(); let (reply, response) = oneshot::channel(); @@ -583,7 +573,6 @@ impl CellHandle { if retained >= limit.saturating_sub(limit / 4) || disk.used() >= disk.capacity().saturating_sub(disk.capacity() / 4) { - self.admission.publication_batch.flush(); return Err(Error::Capacity("publication backlog")); } } diff --git a/crates/cellule-runtime/src/cell/actor/lifecycle/eviction.rs b/crates/cellule-runtime/src/cell/actor/lifecycle/eviction.rs index bce92451..41aeb16a 100644 --- a/crates/cellule-runtime/src/cell/actor/lifecycle/eviction.rs +++ b/crates/cellule-runtime/src/cell/actor/lifecycle/eviction.rs @@ -117,7 +117,7 @@ pub(in crate::cell::actor) fn begin_idle_cell_eviction( } return false; } - active.admission.begin_drain(); + active.admission.draining.store(true, Ordering::Release); active.admission.requests.close(); active.admission.bytes.close(); active.drain = reply.map(DrainReply::Unit); diff --git a/crates/cellule-runtime/src/cell/actor/mod.rs b/crates/cellule-runtime/src/cell/actor/mod.rs index 08099f03..ddfa53b3 100644 --- a/crates/cellule-runtime/src/cell/actor/mod.rs +++ b/crates/cellule-runtime/src/cell/actor/mod.rs @@ -24,7 +24,6 @@ mod acquire; mod acquire_resume; mod acquisition_observer; mod prefix; -mod publication_schedule; mod serving; pub use acquisition_observer::{AcquisitionObservation, AcquisitionObserver}; pub use serving::CellServingObservation; diff --git a/crates/cellule-runtime/src/cell/actor/publication_schedule/mod.rs b/crates/cellule-runtime/src/cell/actor/publication_schedule/mod.rs deleted file mode 100644 index 4fe19aeb..00000000 --- a/crates/cellule-runtime/src/cell/actor/publication_schedule/mod.rs +++ /dev/null @@ -1,83 +0,0 @@ -//! Fleet-proven root cohorts wait outside scarce preparation admission. - -use std::{ - sync::atomic::{AtomicBool, AtomicUsize, Ordering}, - time::{Duration, Instant}, -}; - -use tokio::sync::Notify; - -use super::{CellAdmission, MAX_PENDING_PUBLICATIONS, PendingDurability}; - -const BATCH_CAPTURES: usize = MAX_PENDING_PUBLICATIONS / 2; -const BATCH_AGE: Duration = Duration::from_secs(5); -const FOLLOWER_WAIT: Duration = Duration::from_millis(50); - -#[derive(Default)] -pub(super) struct PublicationBatch { - queued: AtomicUsize, - flush: AtomicBool, - changed: Notify, -} - -impl PublicationBatch { - pub(super) fn queued(&self, count: usize, object_fallback: bool) { - self.queued.store(count, Ordering::Release); - if object_fallback { - self.flush.store(true, Ordering::Release); - } - self.changed.notify_one(); - } - - pub(super) fn flush(&self) { - self.flush.store(true, Ordering::Release); - self.changed.notify_one(); - } - - pub(super) fn reset(&self, count: usize, object_fallback: bool) { - self.queued.store(count, Ordering::Release); - self.flush.store(object_fallback, Ordering::Release); - } - - fn ready(&self, admission: &CellAdmission) -> bool { - admission.draining.load(Ordering::Acquire) - || admission.fenced.load(Ordering::Acquire) - || self.flush.load(Ordering::Acquire) - || self.queued.load(Ordering::Acquire) >= BATCH_CAPTURES - } -} - -pub(super) async fn wait_for_batch( - admission: &CellAdmission, - durability: &PendingDurability, - submitted_at: Instant, -) { - // This is scheduling, never an ACK gate. Only an already-active original - // Fleet lane can defer a root, and this particular complete capture must - // first obtain its follower proof. Failure/stall starts ordinary fallback. - if !durability.fleet_active() - || admission.publication_batch.ready(admission) - || !matches!( - tokio::time::timeout(FOLLOWER_WAIT, durability.prove_fleet()).await, - Ok(Ok(())) - ) - { - return; - } - let deadline = submitted_at + BATCH_AGE; - loop { - let changed = admission.publication_batch.changed.notified(); - tokio::pin!(changed); - changed.as_mut().enable(); - if admission.publication_batch.ready(admission) - || !durability.fleet_active() - || Instant::now() >= deadline - { - return; - } - tokio::select! { - () = changed => {}, - () = tokio::time::sleep_until(deadline.into()) => return, - } - } -} diff --git a/crates/cellule-runtime/src/cell/actor/requests.rs b/crates/cellule-runtime/src/cell/actor/requests.rs index 10722c06..c73e9f74 100644 --- a/crates/cellule-runtime/src/cell/actor/requests.rs +++ b/crates/cellule-runtime/src/cell/actor/requests.rs @@ -404,25 +404,10 @@ pub(super) fn start_publication( let generation = active.generation; let effect_id = active.begin_task(CoordinationEffect::Publication); let fleet_deadline = std::time::Instant::now() + FLEET_PUBLICATION_GRACE; - let oldest = active - .publications - .front() - .map(|queued| (queued.submitted_at, queued.durability.clone())); - let batch_admission = Arc::clone(&active.admission); tasks.spawn(async move { // Keep coverage in the bounded Cell queue while waiting. The publisher // token prevents another root or compaction from overtaking admission. - let result = async { - if publisher.publication_saturated() - && let Some((submitted_at, Some(durability))) = oldest - { - publication_schedule::wait_for_batch(&batch_admission, &durability, submitted_at) - .await; - } - publisher.admit_publication().await - } - .await - .map(|replica| { + let result = publisher.admit_publication().await.map(|replica| { Box::new(PublicationAdmission { replica, fleet_deadline, @@ -476,13 +461,6 @@ pub(super) fn start_admitted_publication( active.publisher = Some(*publisher); return; }; - active.admission.publication_batch.reset( - active.publications.len(), - active - .publications - .iter() - .any(|queued| queued.durability.is_none()), - ); let Some(newest) = coverage.last() else { active.finish_task(effect_id, CoordinationEffect::Publication); active.publisher = Some(*publisher); diff --git a/crates/cellule-runtime/src/cell/actor/task.rs b/crates/cellule-runtime/src/cell/actor/task.rs index 6c6069f9..2fbffc8e 100644 --- a/crates/cellule-runtime/src/cell/actor/task.rs +++ b/crates/cellule-runtime/src/cell/actor/task.rs @@ -309,7 +309,7 @@ pub(super) fn start_shutdown_drain( } active.inventory_refreshing = false; active.coordination.step(CoordinationInput::BeginShutdown); - active.admission.begin_drain(); + active.admission.draining.store(true, Ordering::Release); active.admission.requests.close(); active.admission.bytes.close(); if !active.queue.is_empty() { @@ -914,7 +914,7 @@ pub(super) fn handle_message( return; } } - active.admission.begin_drain(); + active.admission.draining.store(true, Ordering::Release); active.admission.requests.close(); active.admission.bytes.close(); active.admission = new_cell_admission(active.admission.owner_fence); diff --git a/crates/cellule-runtime/src/cell/actor/tasks/activation.rs b/crates/cellule-runtime/src/cell/actor/tasks/activation.rs index dcd88104..4fe65acb 100644 --- a/crates/cellule-runtime/src/cell/actor/tasks/activation.rs +++ b/crates/cellule-runtime/src/cell/actor/tasks/activation.rs @@ -51,7 +51,7 @@ pub(super) fn handle_activated( return; } if shutdown.draining { - admission.begin_drain(); + admission.draining.store(true, Ordering::Release); admission.requests.close(); admission.bytes.close(); let _ = reply.send(Err(Error::RuntimeClosed)); diff --git a/crates/cellule-runtime/src/cell/actor/tasks/maintenance_transfer.rs b/crates/cellule-runtime/src/cell/actor/tasks/maintenance_transfer.rs index 89388781..586756a0 100644 --- a/crates/cellule-runtime/src/cell/actor/tasks/maintenance_transfer.rs +++ b/crates/cellule-runtime/src/cell/actor/tasks/maintenance_transfer.rs @@ -94,7 +94,7 @@ pub(super) fn handle( // work and refreshing readiness again. Keep semaphore owners // intact until confirm, so a definite refusal can restore // only native completion on this same exact activation. - active.admission.begin_drain(); + active.admission.draining.store(true, Ordering::Release); state.closing = true; } state.next_check = now + HYDRATION_TICK; diff --git a/crates/cellule-runtime/src/cell/actor/tasks/work.rs b/crates/cellule-runtime/src/cell/actor/tasks/work.rs index 38862a44..994f49de 100644 --- a/crates/cellule-runtime/src/cell/actor/tasks/work.rs +++ b/crates/cellule-runtime/src/cell/actor/tasks/work.rs @@ -98,10 +98,6 @@ pub(super) fn handle_executed( submitted_at: std::time::Instant::now(), proof, }); - active - .admission - .publication_batch - .queued(active.publications.len(), durability.is_none()); if durability.is_some() { active.unpublished_node_logs += 1; unpublished_node_log_bytes.fetch_add(retained_bytes, Ordering::AcqRel); diff --git a/crates/cellule-runtime/src/cell/actor/tests.rs b/crates/cellule-runtime/src/cell/actor/tests.rs index f20bf5fb..502b0377 100644 --- a/crates/cellule-runtime/src/cell/actor/tests.rs +++ b/crates/cellule-runtime/src/cell/actor/tests.rs @@ -1,269 +1,5 @@ use super::{CellRuntimeStats, bounded_u32}; -#[tokio::test(flavor = "multi_thread")] -async fn pressured_fleet_roots_batch_without_hiding_reads_or_delaying_drain() { - use super::*; - use crate::cell::catalog::CellCatalog; - use crate::cell::executor::{HandlerOutcome, MutationIdentity}; - use crate::control::Owner; - use crate::identity::{IncarnationId, NamespaceId, NodeId, RequestId, TenantId}; - use crate::node::log::DurabilityGate; - use crate::node::log_shipper::NodeLogShipper; - use crate::node::log_transport::LocalFollowerTransport; - - let session = SessionId::from_bytes([87; 16]); - let incarnation = IncarnationId::from_bytes([88; 16]); - let member = NodeId::from_bytes([89; 16]); - let target = CellTarget::new( - TenantId::from_bytes([1; 16]), - ApplicationId::from_bytes([3; 16]), - NamespaceId::from_bytes([6; 16]), - b"pressured-fleet-roots", - ) - .unwrap(); - let layout = cellule_ltx::CellStorageLayout::new( - cellule_store::Store::new(Arc::new(object_store::memory::InMemory::new())), - object_store::path::Path::from("pressured-roots"), - *target.application().as_bytes(), - ); - let limits = cellule_ltx::Limits::default(); - let replica = cellule_ltx::CellReplica::new( - layout.clone(), - *target.cell_id().as_bytes(), - *incarnation.as_bytes(), - limits, - ) - .unwrap(); - let directory = tempfile::TempDir::new().unwrap(); - let follower = crate::FollowerStore::open( - directory.path().join("follower"), - limits, - cellule_ltx::DiskBudget::new(1 << 30), - ) - .unwrap(); - let transport = Arc::new(LocalFollowerTransport::new(member, follower)); - let gate = DurabilityGate::new(session, NodeId::from_bytes([90; 16]), 1, [member]).unwrap(); - let shipper = NodeLogShipper::new(gate.clone(), transport.clone(), limits).unwrap(); - let node_lease = NodeLeaseGuard::new(0, 60_000).unwrap(); - let runtime = CellRuntime::new_with_replica_host_requiring_node_lease( - SqlWorkerPool::new(1, 1).unwrap(), - 8 << 20, - session, - cellule_ltx::Host::default(), - ) - .unwrap(); - runtime.install_node_lease(node_lease.clone()).unwrap(); - runtime - .install_node_durability( - target.application(), - Arc::new(NodeDurability::new( - gate, - shipper, - Arc::new(BatchNodeAuthority), - transport, - node_lease, - )), - ) - .unwrap(); - let catalog = CellCatalog::new(layout.clone(), target.tenant()); - let code = Digest::from_bytes([91; 32]); - let proof = catalog - .provision(CatalogEntry::new(&target, CatalogRole::Application, code, 1).unwrap()) - .await - .unwrap(); - let authority = CellAuthority::new(layout); - let observed = authority - .create_initial( - &proof, - incarnation, - Owner { - session, - endpoint: "https://node.internal".into(), - }, - ) - .await - .unwrap(); - let handle = runtime - .bootstrap( - proof, - replica.clone(), - authority.clone(), - observed, - directory.path().join("cell.sqlite"), - |tx| { - tx.execute_batch( - "CREATE TABLE counter(value INTEGER); INSERT INTO counter VALUES (0)", - )?; - Ok(()) - }, - ) - .await - .unwrap(); - let identity = |id| MutationIdentity { - request_id: RequestId::from_bytes([id; 16]), - issued_at_ms: 10, - expires_at_ms: 60_000, - }; - let mut publications = runtime.subscribe_publications(); - let mut warmup_slots = Vec::new(); - for _ in 0..cellule_ltx::SHARED_PUBLICATION_ROWS { - warmup_slots.push(runtime.inner.shared_publication.admit().await.unwrap()); - } - let first = handle - .execute(identity(1), code, 20, 1_024, 1_024, |tx| { - tx.execute("UPDATE counter SET value = value + 1", [])?; - Ok(HandlerOutcome::Success(vec![1])) - }) - .await - .unwrap(); - assert_eq!(first.commit_sequence(), 1); - assert!( - runtime - .node_durability() - .unwrap() - .1 - .progress() - .unwrap() - .fleet_active - ); - drop(warmup_slots); - tokio::time::timeout(std::time::Duration::from_secs(5), publications.recv()) - .await - .unwrap() - .unwrap(); - tokio::time::timeout(std::time::Duration::from_secs(5), async { - loop { - let control = authority.load(target.cell_id()).await.unwrap().unwrap(); - if control.value().ltx_root().unwrap().commit_sequence == 1 { - break; - } - tokio::task::yield_now().await; - } - }) - .await - .unwrap(); - - // Model the bounded node lane's complete occupancy without retaining any - // synthetic captures. These are its real original preparation permits. - let mut occupied = Vec::new(); - for _ in 0..cellule_ltx::SHARED_PUBLICATION_ROWS { - occupied.push(runtime.inner.shared_publication.admit().await.unwrap()); - } - let mut acknowledged = Vec::new(); - for id in 2_u8..=12 { - let outcome = handle - .execute(identity(id), code, 20, 1_024, 1_024, move |tx| { - tx.execute("UPDATE counter SET value = value + 1", [])?; - Ok(HandlerOutcome::Success(vec![id])) - }) - .await - .unwrap(); - acknowledged.push(outcome); - } - drop(occupied); - // Once pressure clears, an already-proven cohort still owns its original - // bounded age. It does not immediately turn into another tiny root. - tokio::time::sleep(std::time::Duration::from_millis(200)).await; - let current = authority.load(target.cell_id()).await.unwrap().unwrap(); - assert_eq!(current.value().ltx_root().unwrap().commit_sequence, 1); - assert_eq!( - handle - .query(1_024, 1_024, |db| { - let value: i64 = db.query_row("SELECT value FROM counter", [], |row| row.get(0))?; - Ok(value.to_le_bytes().to_vec()) - }) - .await - .unwrap(), - 12_i64.to_le_bytes() - ); - for (id, expected) in (2_u8..=12).zip(acknowledged) { - assert_eq!( - handle - .execute(identity(id), code, 20, 1_024, 1_024, |_| { - panic!("exact retry must not execute SQL") - }) - .await - .unwrap(), - expected - ); - } - // Drain must wake the original delayed task rather than wait its five- - // second age. It still selects and releases the complete captured prefix. - tokio::time::timeout(std::time::Duration::from_secs(2), handle.drain()) - .await - .unwrap() - .unwrap(); - runtime.shutdown().await.unwrap(); - let root = authority - .load(target.cell_id()) - .await - .unwrap() - .unwrap() - .value() - .ltx_root() - .unwrap(); - assert_eq!(root.commit_sequence, 12); - let restored = directory.path().join("cold.sqlite"); - replica - .open_root(&root) - .await - .unwrap() - .restore(&restored) - .await - .unwrap(); - let db = cellule_ltx::rusqlite::Connection::open(restored).unwrap(); - assert_eq!( - db.query_row("SELECT value FROM counter", [], |row| row.get::<_, i64>(0)) - .unwrap(), - 12 - ); - assert_eq!( - db.query_row("SELECT count(*) FROM sys_requests", [], |row| row - .get::<_, i64>(0)) - .unwrap(), - 12 - ); - for id in 1_u8..=12 { - let actual = db - .query_row( - "SELECT operation_digest, result, commit_sequence FROM sys_requests WHERE request_id = ?1", - [RequestId::from_bytes([id; 16]).as_bytes().as_slice()], - |row| Ok((row.get::<_, Vec>(0)?, row.get::<_, Vec>(1)?, row.get::<_, i64>(2)?)), - ) - .unwrap(); - assert_eq!(actual, (code.as_bytes().to_vec(), vec![id], i64::from(id))); - } - let stats = runtime.stats(); - assert_eq!(stats.active_cells(), 0); - assert_eq!(stats.retained_bytes(), 0); -} - -struct BatchNodeAuthority; - -impl crate::node::durability::NodeLogAuthority for BatchNodeAuthority { - fn activate<'a>( - &'a self, - _epoch: u64, - ) -> futures_util::future::BoxFuture<'a, crate::Result<()>> { - Box::pin(async { Ok(()) }) - } - - fn advance_coverage<'a>( - &'a self, - _epoch: u64, - _through: u64, - ) -> futures_util::future::BoxFuture<'a, crate::Result<()>> { - Box::pin(async { Ok(()) }) - } - - fn close<'a>( - &'a self, - _retirement: &'a crate::node::log::NodeLogRetirementObservation, - ) -> futures_util::future::BoxFuture<'a, crate::Result<()>> { - Box::pin(async { Ok(()) }) - } -} - #[tokio::test] async fn empty_runtime_miss_does_not_wait_for_the_dispatcher() { use super::*; diff --git a/crates/cellule-runtime/src/publication/mod.rs b/crates/cellule-runtime/src/publication/mod.rs index 8c37df9d..55b5f827 100644 --- a/crates/cellule-runtime/src/publication/mod.rs +++ b/crates/cellule-runtime/src/publication/mod.rs @@ -161,12 +161,6 @@ impl CellPublisher { .await } - pub(crate) fn publication_saturated(&self) -> bool { - self.shared_publication - .as_ref() - .is_some_and(|publication| publication.saturated()) - } - pub(crate) fn durability_submitter(&self) -> CellDurabilitySubmitter { let control = self.observed.value(); CellDurabilitySubmitter { @@ -1157,12 +1151,6 @@ pub(crate) struct PendingDurability { } impl PendingDurability { - pub(crate) fn fleet_active(&self) -> bool { - self.durability - .progress() - .is_ok_and(|progress| progress.fleet_active) - } - pub(crate) async fn prove(&self) -> Result { let proof = self.durability.prove(self.ticket).await?; if proof.source() == crate::node::log::DurabilitySource::Bundle { diff --git a/crates/cellule-runtime/src/publication/shared/mod.rs b/crates/cellule-runtime/src/publication/shared/mod.rs index be8b3f81..84cc82e0 100644 --- a/crates/cellule-runtime/src/publication/shared/mod.rs +++ b/crates/cellule-runtime/src/publication/shared/mod.rs @@ -56,10 +56,6 @@ pub(crate) struct SharedPublication { } impl SharedPublication { - pub(crate) fn saturated(&self) -> bool { - self.slots.available_permits() == 0 - } - pub(crate) fn new( resources: ResourceLedger, telemetry: crate::fleet::telemetry::CellTelemetryHandle, From 61f01363696eb221488351ed223b73e824ae62c1 Mon Sep 17 00:00:00 2001 From: forhappy Date: Thu, 8 Oct 2026 02:15:39 -0700 Subject: [PATCH 031/102] Record matched Fleet root-delay regression and rollback --- docs/pr67-fleet-root-delay-measurement.md | 152 ++++++++++++++++++ docs/pr67-sparse-root-coverage-measurement.md | 3 + 2 files changed, 155 insertions(+) create mode 100644 docs/pr67-fleet-root-delay-measurement.md diff --git a/docs/pr67-fleet-root-delay-measurement.md b/docs/pr67-fleet-root-delay-measurement.md new file mode 100644 index 00000000..aebdfb9d --- /dev/null +++ b/docs/pr67-fleet-root-delay-measurement.md @@ -0,0 +1,152 @@ +# PR67 Fleet root-delay measurement + +**The root-delay experiment was rejected and reverted.** One matched +five-minute window completed 350.62 writes/s versus 403.73 for the retained +implementation, a 13.2% observed regression. Successful scheduled p99 improved, +but request p99 worsened, root density did not improve, and both Cellule runs +failed audit and drain. This is not a performance-parity delivery or a +sustainable-capacity result. + +## Measured change and rollback + +Experiment `bb2f08986cb16ff0c0d477226767670f3fcd0a4f` deferred an already +follower-proven Cell root outside a saturated shared preparation lane. It +requested preparation after 32 physical captures, five seconds from the oldest +capture, or a fallback/backlog/drain/fence signal. The existing proof gate, +capture limits and publication retry deadline remained unchanged. Ordinary +bundle-based application ACKs remained disabled. + +Revert `ddb030c776e5d61a75b76d54e8610120be781df0` restores exactly the complete +`e900bfae21fba2d073316707b18f4a80d292d7d2` tree. That delivery's production source +matches measured baseline `b1856728984781cee5398f8d7185fb88fde23993`; their +differences are documentation only. The additional report in this delivery does +not change the measured application behavior. The rejected experiment and its +failed evidence remain available in history and outside Git. + +One overloaded pair cannot establish repeatable causal attribution. Removing +an unverified delay avoids adding reader lag and lifecycle complexity without +measured root-density or throughput benefit. + +## Matched workload + +Runs completed on 2026-10-08 UTC, in baseline/candidate/celld order. The baseline +reuses the verified `b185672` release binary; the experiment is a fresh release +build. celld v0.6.1 is pinned to +`f2bf648663a610eefde71f3547ad61e9b896b1f0`. Source, image, fixture and binary +identities are retained; clients and auditors are byte-identical. + +Each arm used 1,000 uniform Cells, 96-byte values, SQL INSERT plus SELECT and a +two-hour durable request/result ledger, 128 clients, a 128-offer queue, +30-second warmup and a 300-second window offering 15,000 writes/s. There was +**one repetition** and no separate read load. Bucket, read-only and mixed +performance were not rerun. This workload does not validate the separate +laptop's reported 15K write TPS or the 2,000-Cell serving-node target. + +All roles shared one ARM64 Linux VM with 8 CPUs and nominal 16 GiB RAM. Nodes +retained 4 GiB tmpfs ceilings and Cellule's 64 MiB retained-memory/1 GiB +managed-disk budgets. RustFS retained 2 CPUs and an 8 GiB memory ceiling in the +separate provider-headroom diagnostic. The original qualification profiles and +deadlines were unchanged; their prior failed 2 GiB provider run remains in the +[earlier report](pr67-write-measurement-4957985.md). Another VM was running on the +physical host. Dedicated serving-node capacity and physical-device durability +are unqualified. + +## Actual completed windows + +TPS counts successful logical writes completed inside the 300-second window. +Successful percentiles use exact nearest-rank client-journal values, including +late successful responses. Scheduled latency includes waiting from the offered +time; request latency starts when the client sends the request. Fast errors and +queue drops are excluded from these successful percentiles. + +| Metric | Retained Cellule baseline | Rejected experiment | celld | +| --- | ---: | ---: | ---: | +| Completed writes/s | 403.73 | 350.62 | 1,107.42 | +| Successful writes inside window | 121,120 | 105,187 | 332,226 | +| Successful scheduled p50 ms | 107.15 | 111.80 | 67.01 | +| Successful scheduled p95 ms | 427.97 | 359.75 | 153.97 | +| Successful scheduled p99 ms | 1,030.44 | 767.49 | 562.32 | +| Successful request p99 ms | 499.85 | 600.02 | 255.69 | +| Window errors | 3,104,546 | 3,300,026 | 1,354,855 | +| Queue drops | 1,274,320 | 1,094,700 | 2,812,919 | +| Late successful responses | 14 | 87 | 0 | + +The experiment's scheduled p99 improved 25.5%, while request p99 worsened +20.0% and median scheduled latency worsened 4.3%. It achieved 31.7% of celld's +observed completion rate. **All three rates are failing-run observations.** +The previous window of the unchanged baseline measured 378.88 TPS versus 403.73 +in this window; +the [previous comparison](pr67-sparse-root-coverage-measurement.md) remains +separate. No repeatable gain, healthy celld capacity or parity claim follows. + +## Cost and publication pressure + +| Window metric | Baseline | Experiment | +| --- | ---: | ---: | +| Selected Cell roots | 54,930 | 48,922 | +| Materialized commits | 111,265 | 96,322 | +| Materialized commits per root | 2.026 | 1.969 | +| All-provider successful PUTs per completed command | 1.971 | 1.965 | +| Node-authority-family successful PUTs | 796 | 910 | +| End oldest unpublished age ms | 189,537 | 203,144 | +| Peak sampled retained-capture bytes | 69,813,120 | 71,136,510 | + +These storage API ratios include all serving nodes, exclude trailing +publication and SDK-internal retries, and have outstanding work at the window +boundaries. They are not total lifecycle cost. Node-authority-family counts +also include coordination. The delay did not deliver denser materialized +roots or a substantial reduction in PUTs per command. + +Both Cellule runs remained Fleet-active and nonrotating in all six samples. +Both fail the late-window publication-age slope gate; the experiment also +fails the debt and pending-publication slopes. Its final issued/follower-proven +frontiers were 129,234/129,174 versus tiered 67,430. This gap includes +already-rooted sparse completions and is **not a count of missing or lost +writes**. + +## Audit, drain and local capacity failures + +| Case | Warm ACK read/exact-retry audit | Drain and cold recovery | +| --- | --- | --- | +| Baseline | 143,024 ACKs checked; 141,913 retries passed; 1,111 errors | Owner missed 120-second deadline; forced cleanup; cold not reached | +| Experiment | 124,293 ACKs checked; 123,502 retries passed; 791 errors | Owner missed 120-second deadline; forced cleanup; cold not reached | +| celld | 476,269 ACKs checked; no retries passed; 476,269 errors | Cleanup exited zero; cold not reached after audit failure | + +Cellule's retained audit examples are HTTP 503; celld's are HTTP 500. They do +not classify every failure or establish data loss. Unlike its previous run, +celld's three nodes exited zero during cleanup and had false OOM flags. +Its owner log repeatedly reports WAL capture failures with `database or disk +is full` and `No space left on device`. Local state used the configured 4 GiB +tmpfs. A direct post-window filesystem read failed because the owner had +already stopped; no exact occupancy snapshot was captured. This invalidates a +healthy-capacity interpretation without establishing the complete failure +chain. Live provider samples had byte/inode headroom; required post-audit and +cold-lifecycle snapshots remain missing because audits failed. + +## Verification and remaining architecture + +The rejected frozen source passed all 11 contributor checks, including 1,942 +top-level workspace cases and doctests, one nested subprocess, 38 ignored and +60 local LTX cases. Its actor regression verified Fleet ACKs, owner reads, +exact retries, drain wakeup and all 12 outcomes after canonical cold restore; +it failed against the prior source as expected. That fixture simulates occupied +preparation permits and does not qualify distributed overload or hardware +durability. The revert restores the previously verified production tree. + +Raw evidence stays outside Git under `cellule-write-perf-8ad1`: the plan, +immutable builds, every client/ACK journal, failed audits, logs and retained +Docker volumes. Normalized results are +`fleet-root-batch-bb2f089-results.json`; independent streaming verification is +`fleet-root-batch-bb2f089-evidence-index.json`. Verification reconciles offered, +attempted, failed and successful writes, exact successful percentiles, +acknowledgement manifests, measured binaries and 535 artifact hashes. The +dedicated benchmark VM is stopped; the user's default VM and Docker context +remain unchanged. + +The remaining work is verified shared selection in actor command/read/retry +paths with exact capture release, admitted asynchronous root materialization, +complete issued-range drain/recovery and safe cross-Cell collection. Per-Cell +delay is not a substitute for that architecture. The +[design's exit gates](../crates/cellule-runtime/docs/write-performance-design.md) +still require three matched repetitions, zero errors/drops, stable publication, +all-ACK cold recovery, complete drain and read guardrails. diff --git a/docs/pr67-sparse-root-coverage-measurement.md b/docs/pr67-sparse-root-coverage-measurement.md index ab360308..6fff5423 100644 --- a/docs/pr67-sparse-root-coverage-measurement.md +++ b/docs/pr67-sparse-root-coverage-measurement.md @@ -1,5 +1,8 @@ # PR67 sparse-root coverage measurement +For the subsequent root-delay experiment and its rollback, see the +[new matched measurement](pr67-fleet-root-delay-measurement.md). + **Performance parity is not delivered.** The live-path change observed 378.88 successful writes/s versus 282.43 before it in one matched five-minute run. Successful-write p99 improved, but median latency worsened, overload errors From d4bec27351310d1a96ca0579d2115edfa15969af Mon Sep 17 00:00:00 2001 From: forhappy Date: Thu, 8 Oct 2026 02:43:20 -0700 Subject: [PATCH 032/102] Retain complete native captures in an ordered publication feed --- .../docs/failover-and-followers.md | 15 ++ .../src/node/durability/mod.rs | 8 + .../src/node/log_shipper/mod.rs | 66 +++++++- .../src/node/log_shipper/publication/mod.rs | 117 +++++++++++++ .../src/node/log_shipper/tests.rs | 158 +++++++++++++++++- 5 files changed, 354 insertions(+), 10 deletions(-) create mode 100644 crates/cellule-runtime/src/node/log_shipper/publication/mod.rs diff --git a/crates/cellule-runtime/docs/failover-and-followers.md b/crates/cellule-runtime/docs/failover-and-followers.md index b94b6ba9..0230d15f 100644 --- a/crates/cellule-runtime/docs/failover-and-followers.md +++ b/crates/cellule-runtime/docs/failover-and-followers.md @@ -291,6 +291,21 @@ set, activation bit, contiguous object watermark, and renewable recovery claim. failure, while its tickets remain eligible for object proof and covered rotation. +`NodeDurability::take_publication_feed` installs one ordered consumer before +the original epoch issues any frames. Each `AssignedCapture` retains its exact +complete assignment and verified frames under the shipper's existing byte +admission. The 512-submission queue reserves space before sequence issuance; +cancellation or a full queue cannot leave an issued gap. A capture larger than +64 frames remains one complete witness even when follower transport splits it. +Shutdown closes admission, wakes blocked producers and lets the feed drain +accepted captures. The host must join selection or verified fallback for all +of them before retiring the epoch. + +This feed grants no publication authority or response proof. Ordinary actor +bundle ACKs remain disabled until shared selection, visibility, capture release +and complete issued-range drain are integrated. The example application does +not install the feed yet; this API alone is not a measured throughput gain. + **Ensemble directory.** It filters live peers by protocol, pressure, and the exact shared-disk capacity advertised by their follower stores: diff --git a/crates/cellule-runtime/src/node/durability/mod.rs b/crates/cellule-runtime/src/node/durability/mod.rs index 0041e3cc..a336ae2f 100644 --- a/crates/cellule-runtime/src/node/durability/mod.rs +++ b/crates/cellule-runtime/src/node/durability/mod.rs @@ -175,6 +175,14 @@ impl NodeDurability { self.gate.progress() } + /// Installs this epoch's sole ordered publication consumer before issuance. + /// The original host must join selection/fallback for every accepted capture + /// before retiring this epoch. Receiving captures grants no durability proof. + pub fn take_publication_feed(&self) -> Result { + self.node_lease.check()?; + self.shipper.take_publication_feed() + } + /// Creates one node-log durability epoch over its gate, shipper, authority, /// transport, and node lease. #[must_use] diff --git a/crates/cellule-runtime/src/node/log_shipper/mod.rs b/crates/cellule-runtime/src/node/log_shipper/mod.rs index 520f3c3f..0a9e978c 100644 --- a/crates/cellule-runtime/src/node/log_shipper/mod.rs +++ b/crates/cellule-runtime/src/node/log_shipper/mod.rs @@ -11,6 +11,9 @@ use crate::node::log::{CommitTicket, DurabilityGate}; use crate::node::log_transport::{AppendRequest, NodeLogTransport}; use crate::{Error, Result}; +mod publication; +pub use publication::{AssignedCapture, NodePublicationFeed}; + const MAX_BATCH_FRAMES: usize = 64; const MAX_QUEUED_SUBMISSIONS: usize = 512; const NODE_FRAME_HEADER_BYTES: u64 = 240; @@ -195,6 +198,8 @@ pub struct NodeLogShipper { max_outstanding_bytes: u64, gate: DurabilityGate, limits: cellule_ltx::Limits, + publication: publication::PublicationState, + stopping: tokio::sync::watch::Sender, } impl NodeLogShipper { @@ -236,6 +241,7 @@ impl NodeLogShipper { let (sender, receiver) = mpsc::channel(MAX_QUEUED_SUBMISSIONS); let bytes = Arc::new(Semaphore::new(permits)); let worker_gate = gate.clone(); + let (stopping, _) = tokio::sync::watch::channel(false); let worker = runtime.spawn(run_shipper( receiver, worker_gate, @@ -247,6 +253,7 @@ impl NodeLogShipper { batch_bytes, telemetry.clone(), interval, + stopping.clone(), )); Ok(Self { sender: std::sync::Mutex::new(Some(sender)), @@ -256,9 +263,25 @@ impl NodeLogShipper { max_outstanding_bytes: batch_bytes, gate, limits, + publication: publication::PublicationState::default(), + stopping, }) } + /// Installs the sole ordered publication consumer before any native issuance. + /// The host must own and join this consumer with the original epoch. Taking + /// the feed does not select objects, authorize responses or activate Fleet. + pub fn take_publication_feed(&self) -> Result { + let _ordered = self + .order + .try_lock() + .map_err(|_| Error::PendingPublication)?; + if self.gate.issued_through() != 0 || *self.stopping.borrow() { + return Err(Error::PendingPublication); + } + self.publication.take_feed(self.stopping.subscribe()) + } + pub(crate) fn validate_limits(limits: cellule_ltx::Limits) -> Result<(u64, usize)> { let batch_bytes = limits .max_capture_bytes @@ -327,6 +350,12 @@ impl NodeLogShipper { // admission. This lane only patches exclusively owned envelopes and // atomically commits their consecutive ticket before enqueueing. let _ordered = self.order.lock().await; + // Reserve both consumers before committing a sequence. Cancellation or + // a full publication queue therefore cannot leave an unselectable gap. + let publication = self.publication.reserve(&self.stopping).await?; + if *self.stopping.borrow() { + return Err(Error::RuntimeClosed); + } let ticket = self.gate.preview(frame_count)?; let encoded = loaded.encode(ticket)?; let assignment = self @@ -336,6 +365,13 @@ impl NodeLogShipper { let reservation = Arc::new(OutstandingBytes { _permit: reservation, }); + if let Some(publication) = publication { + publication.send(AssignedCapture::new( + assignment, + encoded.clone(), + Arc::clone(&reservation), + )); + } let frames = encoded .into_iter() .enumerate() @@ -351,6 +387,8 @@ impl NodeLogShipper { /// Closes admission and drains every accepted frame to the current epoch. pub async fn shutdown(&self) -> Result<()> { + self.stopping.send_replace(true); + let publication_closed = self.publication.close(); self.bytes.close(); self.sender .lock() @@ -361,16 +399,18 @@ impl NodeLogShipper { .lock() .map_err(|_| Error::Node("node-log shipper lock poisoned"))? .take(); - if let Some(worker) = worker { - worker.await.map_err(Error::FollowerWorkerJoin)?; - } + let joined = match worker { + Some(worker) => worker.await.map_err(Error::FollowerWorkerJoin), + None => Ok(()), + }; self.gate.stop_shipping(); - Ok(()) + publication_closed.and(joined) } } impl Drop for NodeLogShipper { fn drop(&mut self) { + self.stopping.send_replace(true); self.gate.stop_shipping(); self.bytes.close(); } @@ -405,6 +445,7 @@ async fn run_shipper( max_batch_bytes: u64, telemetry: crate::fleet::telemetry::CellTelemetryHandle, interval: Duration, + stopping: tokio::sync::watch::Sender, ) { let mut pending = VecDeque::::new(); let mut closed = false; @@ -412,12 +453,14 @@ async fn run_shipper( if pending.is_empty() { if closed { bytes.close(); + stopping.send_replace(true); return; } match receiver.recv().await { Some(submission) => pending.extend(submission.frames), None => { bytes.close(); + stopping.send_replace(true); return; } } @@ -432,18 +475,18 @@ async fn run_shipper( break; }; let Some(next_bytes) = batch_bytes.checked_add(next.encoded.len() as u64) else { - stop_shipper(&gate, &bytes); + stop_shipper(&gate, &bytes, &stopping); return; }; if !batch.is_empty() && next_bytes > max_batch_bytes { break; } if next_bytes > max_batch_bytes { - stop_shipper(&gate, &bytes); + stop_shipper(&gate, &bytes, &stopping); return; } let Some(next) = pending.pop_front() else { - stop_shipper(&gate, &bytes); + stop_shipper(&gate, &bytes, &stopping); return; }; batch_bytes = next_bytes; @@ -484,14 +527,19 @@ async fn run_shipper( .await; telemetry.node_log_append(result.is_ok(), append_bytes); if result.is_err() { - stop_shipper(&gate, &bytes); + stop_shipper(&gate, &bytes, &stopping); receiver.close(); return; } } } -fn stop_shipper(gate: &DurabilityGate, bytes: &Semaphore) { +fn stop_shipper( + gate: &DurabilityGate, + bytes: &Semaphore, + stopping: &tokio::sync::watch::Sender, +) { + stopping.send_replace(true); gate.stop_shipping(); bytes.close(); } diff --git a/crates/cellule-runtime/src/node/log_shipper/publication/mod.rs b/crates/cellule-runtime/src/node/log_shipper/publication/mod.rs new file mode 100644 index 00000000..bb4e6526 --- /dev/null +++ b/crates/cellule-runtime/src/node/log_shipper/publication/mod.rs @@ -0,0 +1,117 @@ +//! Complete original captures retained by the same bounded native issuance lane. + +use super::*; +use std::sync::{Mutex, OnceLock}; +use tokio::sync::watch; + +/// One complete captured assignment, in the original node-log order. +/// +/// The native lane constructed and verified these frames before issuance. This +/// value retains the existing byte admission until dropped. It is a proposal, +/// never selected authority or an ACK, and must not be split into prefix proofs. +pub struct AssignedCapture { + assignment: crate::node::log::AssignedCommitRange, + frames: Vec, + _reservation: Arc, +} + +impl AssignedCapture { + pub(super) fn new( + assignment: crate::node::log::AssignedCommitRange, + frames: Vec, + reservation: Arc, + ) -> Self { + Self { + assignment, + frames, + _reservation: reservation, + } + } + + /// Exact original complete-capture witness accepted by bundle selection. + pub const fn assignment(&self) -> crate::node::log::AssignedCommitRange { + self.assignment + } + + /// Complete verified frames. Borrow them while retaining this admission; + /// copies retained beyond it require the consumer's separate accounting. + pub fn frames(&self) -> &[cellule_ltx::VerifiedNodeFrame] { + &self.frames + } +} + +/// Sole ordered consumer for this original native epoch's publication work. +/// +/// Complete captures share the shipper's outstanding-byte limit and a queue of +/// at most 512 submissions. Receiving does not release admission: the consumer +/// retains each value through joined selection or verified object fallback. +/// A slow consumer applies backpressure before new sequence issuance. +pub struct NodePublicationFeed { + receiver: mpsc::Receiver, + stopping: watch::Receiver, +} + +impl NodePublicationFeed { + /// Receives the next complete capture, including accepted work after close. + /// None means all producers closed and every queued capture was received. + pub async fn recv(&mut self) -> Option { + if *self.stopping.borrow() { + self.receiver.close(); + return self.receiver.recv().await; + } + tokio::select! { + capture = self.receiver.recv() => capture, + _ = self.stopping.changed() => { + self.receiver.close(); + self.receiver.recv().await + } + } + } +} + +#[derive(Default)] +pub(super) struct PublicationState { + sender: OnceLock>>>, +} + +impl PublicationState { + pub(super) fn take_feed(&self, stopping: watch::Receiver) -> Result { + let (sender, receiver) = mpsc::channel(MAX_QUEUED_SUBMISSIONS); + self.sender + .set(Mutex::new(Some(sender))) + .map_err(|_| Error::Node("node publication feed already installed"))?; + Ok(NodePublicationFeed { receiver, stopping }) + } + + pub(super) async fn reserve( + &self, + stopping: &watch::Sender, + ) -> Result>> { + let Some(sender) = self.sender.get() else { + return Ok(None); + }; + let sender = sender + .lock() + .map_err(|_| Error::Node("node publication feed lock poisoned"))? + .clone() + .ok_or(Error::RuntimeClosed)?; + let mut stopping = stopping.subscribe(); + if *stopping.borrow() { + return Err(Error::RuntimeClosed); + } + tokio::select! { + permit = sender.reserve_owned() => permit.map(Some).map_err(|_| Error::RuntimeClosed), + _ = stopping.changed() => Err(Error::RuntimeClosed), + } + } + + pub(super) fn close(&self) -> Result<()> { + if let Some(sender) = self.sender.get() { + sender + .lock() + .map_err(|_| Error::Node("node publication feed lock poisoned"))? + .take(); + } + Ok(()) + } +} diff --git a/crates/cellule-runtime/src/node/log_shipper/tests.rs b/crates/cellule-runtime/src/node/log_shipper/tests.rs index 78e3bf82..a6c26aa5 100644 --- a/crates/cellule-runtime/src/node/log_shipper/tests.rs +++ b/crates/cellule-runtime/src/node/log_shipper/tests.rs @@ -335,12 +335,33 @@ async fn splits_large_submission_at_sixty_four_frames() { ) .unwrap(); - let ticket = shipper.submit(submission(&cuts)).await.unwrap(); + let mut publication = shipper.take_publication_feed().unwrap(); + let (ticket, assignment) = shipper.submit_assigned(submission(&cuts)).await.unwrap(); + let capture = publication.recv().await.unwrap(); + assert_eq!(capture.assignment(), assignment); + assert_eq!(capture.frames().len(), 65); + assignment.verify(capture.frames()).unwrap(); + assert!(assignment.verify(&capture.frames()[..64]).is_err()); assert_eq!( gate.prove(ticket).await.unwrap().source(), crate::node::log::DurabilitySource::Fleet ); shipper.shutdown().await.unwrap(); + assert!(publication.recv().await.is_none()); + let retained = capture + .frames() + .iter() + .map(|frame| frame.encoded().len()) + .sum::(); + assert_eq!( + shipper.bytes.available_permits(), + shipper.max_outstanding_bytes as usize - retained + ); + drop(capture); + assert_eq!( + shipper.bytes.available_permits(), + shipper.max_outstanding_bytes as usize + ); assert_eq!(ticket.first_sequence(), 1); assert_eq!(ticket.last_sequence(), 65); @@ -348,6 +369,141 @@ async fn splits_large_submission_at_sixty_four_frames() { assert!(gate.issue(1).is_err()); } +fn publication_submission(cuts: &cellule_ltx::CaptureBatch, index: u64) -> NodeLogSubmission { + let mut cell = [8; 32]; + cell[..8].copy_from_slice(&index.to_le_bytes()); + NodeLogSubmission::new( + ApplicationId::from_bytes([9; 16]), + CellId::from_bytes(cell), + IncarnationId::from_bytes([7; 16]), + 3, + 4, + cuts, + ) + .unwrap() +} + +#[tokio::test] +async fn cancelled_full_publication_queue_does_not_issue_a_native_gap() { + let (_directory, cuts) = capture(); + let gate = DurabilityGate::new(session(1), node(1), 2, [node(2)]).unwrap(); + gate.activate_fleet().unwrap(); + let shipper = NodeLogShipper::new( + gate.clone(), + Arc::new(RecordingTransport::default()), + cellule_ltx::Limits::default(), + ) + .unwrap(); + let mut feed = shipper.take_publication_feed().unwrap(); + assert!(shipper.take_publication_feed().is_err()); + for index in 0..MAX_QUEUED_SUBMISSIONS as u64 { + shipper + .submit(publication_submission(&cuts, index)) + .await + .unwrap(); + } + assert_eq!(gate.issued_through(), 512); + let blocked = shipper.submit(publication_submission(&cuts, 512)); + assert!( + tokio::time::timeout(Duration::from_millis(25), blocked) + .await + .is_err() + ); + assert_eq!(gate.issued_through(), 512); + let first = feed.recv().await.unwrap(); + assert_eq!(first.assignment().ticket().first_sequence(), 1); + first.assignment().verify(first.frames()).unwrap(); + drop(first); + let next = shipper + .submit(publication_submission(&cuts, 512)) + .await + .unwrap(); + assert_eq!(next.first_sequence(), 513); + gate.wait_followers(next).await.unwrap(); + shipper.shutdown().await.unwrap(); + for expected in 2..=513 { + let capture = feed.recv().await.unwrap(); + assert_eq!(capture.assignment().ticket().first_sequence(), expected); + capture.assignment().verify(capture.frames()).unwrap(); + } + assert!(feed.recv().await.is_none()); + assert_eq!( + shipper.bytes.available_permits(), + shipper.max_outstanding_bytes as usize + ); +} + +#[tokio::test] +async fn shutdown_wakes_full_publication_admission_and_joins_accepted_frames() { + let (_directory, cuts) = capture(); + let gate = DurabilityGate::new(session(1), node(1), 2, [node(2)]).unwrap(); + let transport = Arc::new(RecordingTransport::default()); + let shipper = Arc::new( + NodeLogShipper::new( + gate.clone(), + transport.clone(), + cellule_ltx::Limits::default(), + ) + .unwrap(), + ); + let mut feed = shipper.take_publication_feed().unwrap(); + for index in 0..MAX_QUEUED_SUBMISSIONS as u64 { + shipper + .submit(publication_submission(&cuts, index)) + .await + .unwrap(); + } + let next = publication_submission(&cuts, 512); + let running = Arc::clone(&shipper); + let blocked = tokio::spawn(async move { running.submit(next).await }); + // The ordered lane is held only after native loading, while the full + // publication queue refuses a slot. Observe that actual blocked state. + tokio::time::timeout(Duration::from_secs(1), async { + while shipper.order.try_lock().is_ok() { + tokio::task::yield_now().await; + } + }) + .await + .unwrap(); + tokio::time::timeout(Duration::from_secs(1), shipper.shutdown()) + .await + .unwrap() + .unwrap(); + assert!(matches!(blocked.await.unwrap(), Err(Error::RuntimeClosed))); + assert_eq!(gate.issued_through(), 512); + let mut captures = 0; + while let Some(capture) = feed.recv().await { + captures += 1; + capture.assignment().verify(capture.frames()).unwrap(); + } + assert_eq!(captures, 512); + assert_eq!(transport.batch_sizes(node(2)).iter().sum::(), 512); +} + +#[tokio::test] +async fn dropped_publication_consumer_refuses_issuance_and_cannot_be_replaced() { + let (_directory, cuts) = capture(); + let gate = DurabilityGate::new(session(1), node(1), 2, [node(2)]).unwrap(); + let shipper = NodeLogShipper::new( + gate.clone(), + Arc::new(RecordingTransport::default()), + cellule_ltx::Limits::default(), + ) + .unwrap(); + drop(shipper.take_publication_feed().unwrap()); + assert!(matches!( + shipper.submit(submission(&cuts)).await, + Err(Error::RuntimeClosed) + )); + assert_eq!(gate.issued_through(), 0); + assert!(shipper.take_publication_feed().is_err()); + shipper.shutdown().await.unwrap(); + assert_eq!( + shipper.bytes.available_permits(), + shipper.max_outstanding_bytes as usize + ); +} + #[tokio::test] async fn covered_queued_prefix_keeps_the_uncovered_suffix_fleet_durable() { let (_directory, cuts) = capture(); From 2dfbaa20223fa9473f9a25d4de16cc36f08f6941 Mon Sep 17 00:00:00 2001 From: forhappy Date: Thu, 8 Oct 2026 03:38:20 -0700 Subject: [PATCH 033/102] Record current Fleet TPS and incomplete comparison evidence --- docs/pr67-fleet-root-delay-measurement.md | 4 + docs/pr67-ordered-feed-measurement.md | 125 ++++++++++++++++++++++ 2 files changed, 129 insertions(+) create mode 100644 docs/pr67-ordered-feed-measurement.md diff --git a/docs/pr67-fleet-root-delay-measurement.md b/docs/pr67-fleet-root-delay-measurement.md index aebdfb9d..2c8ce875 100644 --- a/docs/pr67-fleet-root-delay-measurement.md +++ b/docs/pr67-fleet-root-delay-measurement.md @@ -1,5 +1,9 @@ # PR67 Fleet root-delay measurement +The [newer ordered-feed measurement](pr67-ordered-feed-measurement.md) records +the next source's actual Fleet TPS, audit/drain failures and an aborted celld +comparison. These windows remain separate historical evidence. + **The root-delay experiment was rejected and reverted.** One matched five-minute window completed 350.62 writes/s versus 403.73 for the retained implementation, a 13.2% observed regression. Successful scheduled p99 improved, diff --git a/docs/pr67-ordered-feed-measurement.md b/docs/pr67-ordered-feed-measurement.md new file mode 100644 index 00000000..73bba5d2 --- /dev/null +++ b/docs/pr67-ordered-feed-measurement.md @@ -0,0 +1,125 @@ +# PR67 ordered publication feed measurement + +**No write-throughput improvement was demonstrated.** The current source, +`d4bec27351310d1a96ca0579d2115edfa15969af`, completed 369.85 writes/s against +443.41 for the retained `b1856728984781cee5398f8d7185fb88fde23993` baseline. +That is a 16.6% observed decrease in one matched five-minute Fleet pair. +The candidate also failed its warm ACK audit and owner drain. These are +overloaded-run observations, not sustainable capacity or performance parity. +One pair cannot establish a repeatable causal regression. + +## Exact measured implementation + +The candidate adds an ordered, bounded publication feed to the original native +issuance lane. It retains a complete verified capture and assignment under the +existing byte admission, reserves queue space before issuing a sequence, and +wakes blocked producers on shutdown. A 65-frame capture remains one complete +witness when follower transport splits it into 64 and one frames. + +The example application **does not install this feed**. Ordinary actor bundle +ACKs remain disabled; both measured windows recorded zero bundle proofs. +This is a correctness prerequisite for the producer, not an activated shared +selection optimization. It has not delivered a TPS gain. + +## Workload and environment + +The windows ran on 2026-10-08 UTC in baseline/candidate order. Both offered +15,000 writes/s for 300 seconds after 30 seconds of warmup, with 1,000 uniformly +active Cells, 96-byte values, 128 clients and a 128-offer queue. Each command +performs SQL INSERT plus SELECT and retains a two-hour durable request/result +ledger. There was one repetition and no separate read load. This is not the +bounded KV workload used for the separate laptop result. + +All roles shared one ARM64 Linux VM with 8 CPUs and nominal 16 GiB RAM: +owner, two followers, client and RustFS. Node ceilings remained 8 CPUs/16 GiB +with 4 GiB tmpfs each; Cellule retained its 64 MiB memory and 1 GiB managed-disk +budgets. RustFS retained the separate diagnostic's 2-CPU/8-GiB ceiling. Another +VM remained running on the physical host. These resources do not establish +dedicated serving-node capacity or physical-device durability. + +Clients, auditors, workload fixtures, pinned images, loaded runner, Docker host +and point contracts matched. The candidate was a fresh release build without +overlays. Its actual adapted-source cache key and frozen verification source +were independently reconciled with the build manifest and measured binaries. + +## Actual completed windows + +TPS counts successful logical writes completed inside the 300-second window. +Successful latency percentiles are exact nearest-rank values from client +journals. Scheduled latency starts at the offered time; request latency starts +when the request is sent. Errors and queue drops are excluded from successful +percentiles and are reported separately. Neither window had late successes. + +| Metric | Baseline | Candidate | +| --- | ---: | ---: | +| Completed writes/s | 443.41 | 369.85 | +| Successful writes inside window | 133,023 | 110,956 | +| Successful scheduled p50 ms | 100.62 | 104.68 | +| Successful scheduled p95 ms | 496.47 | 370.11 | +| Successful scheduled p99 ms | 1,121.46 | 1,029.88 | +| Successful request p99 ms | 509.98 | 444.17 | +| All-attempt scheduled p99 ms | 251.5 | 166.3 | +| Request errors | 2,722,188 | 3,079,381 | +| Queue drops | 1,644,789 | 1,309,663 | + +Both generated all 4,500,000 scheduled offers. Attempted successes, errors and +queue drops reconcile exactly. Lower successful tail latency alongside fewer +successful writes and more errors does not establish a performance improvement. +Both fail the unchanged delivery, latency and publication-age stability gates. + +## Publication and lifecycle + +| Window observation | Baseline | Candidate | +| --- | ---: | ---: | +| Selected Cell roots | 60,029 | 49,998 | +| Materialized commits per root | 2.207 | 2.165 | +| All-provider successful PUTs per completed command | 2.016 | 1.922 | +| End oldest unpublished age ms | 153,876 | 176,495 | +| Mean owner capture ms | 0.54 | 0.84 | +| Mean owner dirty admission ms | 208.98 | 222.96 | +| Mean owner publication ms | 2,561.30 | 3,013.88 | + +Phase means describe different overlapping cohorts; they cannot be added into a +command critical path. Storage ratios include all serving nodes but exclude +trailing publication and SDK-internal retries. They are not total lifecycle +cost. Publication remains sparse and age grows despite sampled local tmpfs +headroom. Both retain active, nonrotating Fleet in the original epoch. + +The baseline checked all 158,567 ACKs, including seed and warmup, through warm +reads/exact retries and bucket-only cold recovery with zero audit errors. Its +fleet drain took 9.01 seconds. It still fails performance qualification. + +The candidate checked 128,406 ACKs in the warm audit: 128,160 retries passed +and 246 checks failed with HTTP 503 examples. Its owner missed the 120-second +drain deadline and required forced cleanup. Cold recovery was not reached. +These failures do not by themselves establish data loss; they prevent the +candidate from passing the durability/availability qualification. + +## Aborted comparison and verification + +The six-case plan also included pinned celld v0.6.1 +(`f2bf648663a610eefde71f3547ad61e9b896b1f0`) and Bucket at 2,000 offered writes/s. +During the celld Fleet arm, macOS reported only 104 MiB available, temporary-file +creation failed with ENOSPC, and Docker reported a storage I/O error. That +window was interrupted and excluded. None of the Bucket arms was attempted. +There is **no fresh complete celld comparison, Bucket result, read-only result +or mixed result** in this delivery. Earlier results remain separate in the +[previous measurement](pr67-fleet-root-delay-measurement.md). + +The benchmark VM is stopped; the default VM and Docker context remain +unchanged. Raw evidence stays outside Git under `cellule-write-perf-8ad1`: +`ordered-feed-d4bec27-plan.json`, `ordered-feed-d4bec27-plan-status.json`, +`ordered-feed-d4bec27-results.json`, `ordered-feed-d4bec27-cellule-pair.json` +and `ordered-feed-d4bec27-evidence-index.json`. The independent verifier +reconciles the two completed windows' counts, exact successful percentiles, +ACK manifests and source/binary identities. Its overall status intentionally +fails because the plan is incomplete and the celld window is missing. + +The frozen candidate passed all 11 contributor verification routes, including +the publication-feed cancellation, shutdown and complete-witness regressions. +Passing source checks does not qualify performance. Shared selection in actor +ACK/read/retry paths, exact capture release, admitted asynchronous +materialization, complete issued-range drain/recovery and safe cross-Cell +collection remain necessary. The +[design exit gates](../crates/cellule-runtime/docs/write-performance-design.md) +remain unmet. From 3e436013a7e7500694bfdfe5600fe4a69adb1988 Mon Sep 17 00:00:00 2001 From: forhappy Date: Thu, 8 Oct 2026 11:03:44 -0700 Subject: [PATCH 034/102] Release selected bundle captures through the actor publisher --- .../cellule-axum/examples/sql_metrics/mod.rs | 9 +- .../examples/sql_metrics/tests.rs | 10 + .../docs/write-performance-design.md | 22 +- .../src/cell/actor/admission.rs | 5 +- .../cellule-runtime/src/cell/actor/bundle.rs | 54 ++ .../src/cell/actor/lifecycle/activation.rs | 8 +- crates/cellule-runtime/src/cell/actor/mod.rs | 1 + .../src/cell/actor/requests.rs | 60 +- .../cellule-runtime/src/cell/actor/runtime.rs | 2 + .../src/cell/executor/bundle.rs | 88 +++ .../cellule-runtime/src/cell/executor/mod.rs | 22 +- crates/cellule-runtime/src/cell/worker/mod.rs | 42 ++ crates/cellule-runtime/src/cell/worker/run.rs | 18 + crates/cellule-runtime/src/fleet/resource.rs | 36 ++ crates/cellule-runtime/src/fleet/telemetry.rs | 2 + crates/cellule-runtime/src/node/bundle/mod.rs | 41 ++ .../src/node/bundle/selection.rs | 28 + .../src/node/bundle/tests/actor.rs | 603 ++++++++++++++++++ .../src/node/bundle/tests/mod.rs | 2 + .../src/node/bundle/tests/receipts.rs | 209 ++++++ .../src/node/durability/mod.rs | 202 +++++- crates/cellule-runtime/src/node/log/mod.rs | 36 ++ .../src/node/log_shipper/mod.rs | 32 +- .../src/node/log_shipper/publication/mod.rs | 74 ++- crates/cellule-runtime/src/publication/mod.rs | 141 +++- .../cellule-runtime/src/publication/tests.rs | 10 +- docs/bundle-coverage-implementation.md | 46 +- docs/pr67-bundle-receipt-measurement.md | 126 ++++ 28 files changed, 1877 insertions(+), 52 deletions(-) create mode 100644 crates/cellule-runtime/src/cell/actor/bundle.rs create mode 100644 crates/cellule-runtime/src/cell/executor/bundle.rs create mode 100644 crates/cellule-runtime/src/node/bundle/tests/actor.rs create mode 100644 crates/cellule-runtime/src/node/bundle/tests/receipts.rs create mode 100644 docs/pr67-bundle-receipt-measurement.md diff --git a/crates/cellule-axum/examples/sql_metrics/mod.rs b/crates/cellule-axum/examples/sql_metrics/mod.rs index d5f83a67..ca29483f 100644 --- a/crates/cellule-axum/examples/sql_metrics/mod.rs +++ b/crates/cellule-axum/examples/sql_metrics/mod.rs @@ -29,8 +29,8 @@ pub(super) struct QueryMetrics { storage: storage::Accounting, peer: [Histogram; PeerPhase::COUNT], host_capacity: serde_json::Value, - response_sources: [AtomicU64; 3], - response_elapsed: [Histogram; 3], + response_sources: [AtomicU64; 4], + response_elapsed: [Histogram; 4], response_confirmation: Histogram, proof_wait: [Histogram; 3], submission_sources: [AtomicU64; 4], @@ -253,6 +253,7 @@ impl CellTelemetry for QueryMetrics { CommandResponseSource::Recorded => 0, CommandResponseSource::Fleet => 1, CommandResponseSource::Object => 2, + CommandResponseSource::Bundle => 3, }; self.response_sources[index].fetch_add(1, Ordering::Relaxed); self.response_elapsed[index].observe(elapsed); @@ -411,6 +412,7 @@ impl QueryMetrics { ("response_recorded", &self.response_elapsed[0]), ("response_fleet", &self.response_elapsed[1]), ("response_object", &self.response_elapsed[2]), + ("response_bundle", &self.response_elapsed[3]), ("response_confirmation", &self.response_confirmation), ("proof_fleet", &self.proof_wait[0]), ("proof_object", &self.proof_wait[1]), @@ -496,7 +498,8 @@ impl QueryMetrics { "response_sources": { "recorded": self.response_sources[0].load(Ordering::Relaxed), "fleet": self.response_sources[1].load(Ordering::Relaxed), - "object": self.response_sources[2].load(Ordering::Relaxed) + "object": self.response_sources[2].load(Ordering::Relaxed), + "bundle": self.response_sources[3].load(Ordering::Relaxed) }, "submission_sources": { "fleet": self.submission_sources[0].load(Ordering::Relaxed), diff --git a/crates/cellule-axum/examples/sql_metrics/tests.rs b/crates/cellule-axum/examples/sql_metrics/tests.rs index a7a0e693..6156bebb 100644 --- a/crates/cellule-axum/examples/sql_metrics/tests.rs +++ b/crates/cellule-axum/examples/sql_metrics/tests.rs @@ -37,6 +37,11 @@ fn response_and_proof_timings_keep_ack_latency_separate_from_materialization() { cellule_runtime::node::log::DurabilitySource::Bundle, Duration::from_millis(3), ); + metrics.command_response( + CommandResponseSource::Bundle, + Duration::from_millis(4), + Duration::from_millis(1), + ); let sample = metrics.window_snapshot(); assert_eq!( sample["histograms"]["response_fleet"]["nonzero_buckets"], @@ -52,6 +57,11 @@ fn response_and_proof_timings_keep_ack_latency_separate_from_materialization() { ); assert_eq!(sample["response_sources"]["fleet"], 1); assert_eq!(sample["response_sources"]["object"], 0); + assert_eq!(sample["response_sources"]["bundle"], 1); + assert_eq!( + sample["histograms"]["response_bundle"]["nonzero_buckets"], + serde_json::json!([[40, 1]]) + ); assert_eq!( sample["histograms"]["proof_bundle"]["nonzero_buckets"], serde_json::json!([[30, 1]]) diff --git a/crates/cellule-runtime/docs/write-performance-design.md b/crates/cellule-runtime/docs/write-performance-design.md index 806539ea..41fc6d31 100644 --- a/crates/cellule-runtime/docs/write-performance-design.md +++ b/crates/cellule-runtime/docs/write-performance-design.md @@ -48,9 +48,11 @@ flowchart LR K --> L[Exact root cohort checkpoint] ``` -Uploaded bytes are not authority. Bundle ACKs remain disabled until the actor, -worker, read/retry endpoint, recovery and lifecycle consumers agree on one -opaque complete proof. Do not advance the follower reclamation frontier before +Uploaded bytes are not authority. An explicitly installed original publication +feed can now deliver admitted exact bundle receipts through the actor's existing +command, worker and read/retry gate. The example application's ordinary path +still has no node bundle producer. Production enablement requires the remaining +scheduler, recovery, collection and qualification gates. Do not advance the follower reclamation frontier before failed-owner recovery understands the selected bundle prefix. ## Locator density and checkpoint cost @@ -121,8 +123,18 @@ qualification remain required before enabling actor ACKs. The verified native selector now advances enrolled coverage and selects its bundle in one node CAS. Exact local confirmation performs no second CAS and reports a distinct Bundle proof. Original gate and lease identity remain required; cold proofs grant -reconstruction only. Ordinary actors continue waiting for their root fallback -until command/read/retry visibility and capture release consume that proof. +reconstruction only. An installed feed's original admitted receipt now grants +actor command/read/retry visibility. Before root preparation starts, the same +serialized publisher can consume a complete selected oldest capture prefix: +the worker compares every cut's metadata and body digest with its original +assignment, verifies every live receipt, and only then removes the local files. +Outcomes, proof metadata and publication coordination remain retained until +normal root lineage, exact Cell CAS and joined drain complete. A later unproven +suffix remains hidden. Origin materialization preserves the captured due time +and retries storage errors within the existing publication grace. The bounded +selection opportunity is 100 ms for installed feeds; absent or later coverage +uses the original root path. This is not the fair node materializer scheduler +or the 215-command checkpoint density target. The live root fallback now avoids a node authority mutation when an exact Cell root covers only a sparse range beyond an unpublished native gap. That root diff --git a/crates/cellule-runtime/src/cell/actor/admission.rs b/crates/cellule-runtime/src/cell/actor/admission.rs index a4e0fbcd..9421c1cc 100644 --- a/crates/cellule-runtime/src/cell/actor/admission.rs +++ b/crates/cellule-runtime/src/cell/actor/admission.rs @@ -130,9 +130,12 @@ pub(super) fn send_command_reply( let (source, confirmation) = match command.response_proof { Some((DurabilitySource::Fleet, elapsed)) => (CommandResponseSource::Fleet, elapsed), - Some((DurabilitySource::Object | DurabilitySource::Bundle, elapsed)) => { + Some((DurabilitySource::Object, elapsed)) => { (CommandResponseSource::Object, elapsed) } + Some((DurabilitySource::Bundle, elapsed)) => { + (CommandResponseSource::Bundle, elapsed) + } None => (CommandResponseSource::Recorded, std::time::Duration::ZERO), }; // Final admission may refuse a proven command. Observe at the one diff --git a/crates/cellule-runtime/src/cell/actor/bundle.rs b/crates/cellule-runtime/src/cell/actor/bundle.rs new file mode 100644 index 00000000..a26c29bb --- /dev/null +++ b/crates/cellule-runtime/src/cell/actor/bundle.rs @@ -0,0 +1,54 @@ +//! Bundle selection consumed by the actor's existing serialized publisher. + +use super::*; +use crate::publication::VerifiedBundleCapture; + +// This is only a bounded opportunity for the already installed original feed. +// A stalled producer retains the ordinary root fallback, never a new ACK path. +const SELECTION_OPPORTUNITY: std::time::Duration = std::time::Duration::from_millis(100); + +pub(super) async fn selected_prefix( + publisher: &CellPublisher, + durabilities: &[Option], + commit_sequence: u64, + position: cellule_ltx::Position, +) -> crate::Result>> { + if durabilities.is_empty() + || durabilities.iter().any(|pending| { + pending + .as_ref() + .is_none_or(|pending| !pending.has_bundle_capture()) + }) + { + return Ok(None); + } + let selected = async { + let mut captures = Vec::with_capacity(durabilities.len()); + for pending in durabilities.iter().flatten() { + let capture = pending + .selected_capture() + .await? + .ok_or(Error::Control("bundle selection lost original capture"))?; + captures.push(capture); + } + Ok::<_, Error>(captures) + }; + let captures = match tokio::time::timeout(SELECTION_OPPORTUNITY, selected).await { + Ok(captures) => captures?, + Err(_) => return Ok(None), + }; + let Some(newest) = captures.last() else { + return Ok(None); + }; + let proof = &newest.selected().proof; + // A proof can cover later commands outside this task's retained cohort. + // Keep the root fallback in that case; endpoint equality alone cannot + // authorize pruning captures that this publisher does not own. + if proof.commit_sequence() != commit_sequence || proof.position() != position { + return Ok(None); + } + if publisher.control().value().bundle_binding != Some(proof.binding()) { + return Err(Error::Fenced); + } + Ok(Some(captures)) +} diff --git a/crates/cellule-runtime/src/cell/actor/lifecycle/activation.rs b/crates/cellule-runtime/src/cell/actor/lifecycle/activation.rs index ad1b788d..ee6912c5 100644 --- a/crates/cellule-runtime/src/cell/actor/lifecycle/activation.rs +++ b/crates/cellule-runtime/src/cell/actor/lifecycle/activation.rs @@ -28,7 +28,12 @@ pub(in crate::cell::actor) async fn activate_restored_and_publish( job, ) .await?; - if let Err(error) = publisher.activate().await { + let activation = async { + publisher.activate().await?; + publisher.enroll_bundle().await + } + .await; + if let Err(error) = activation { return match pool.deactivate(cell).await { Ok(()) => Err(error), Err(cleanup) => Err(cleanup), @@ -90,6 +95,7 @@ pub(in crate::cell::actor) async fn bootstrap_and_publish( .await?; pool.confirm_bootstrap_published(cell, bootstrap.cuts) .await?; + publisher.enroll_bundle().await?; Ok(()) } .await; diff --git a/crates/cellule-runtime/src/cell/actor/mod.rs b/crates/cellule-runtime/src/cell/actor/mod.rs index ddfa53b3..bfdcb402 100644 --- a/crates/cellule-runtime/src/cell/actor/mod.rs +++ b/crates/cellule-runtime/src/cell/actor/mod.rs @@ -28,6 +28,7 @@ mod serving; pub use acquisition_observer::{AcquisitionObservation, AcquisitionObserver}; pub use serving::CellServingObservation; mod admission; +mod bundle; mod group; mod lifecycle; mod maintenance; diff --git a/crates/cellule-runtime/src/cell/actor/requests.rs b/crates/cellule-runtime/src/cell/actor/requests.rs index c73e9f74..a31b3247 100644 --- a/crates/cellule-runtime/src/cell/actor/requests.rs +++ b/crates/cellule-runtime/src/cell/actor/requests.rs @@ -530,7 +530,7 @@ pub(super) fn start_admitted_publication( } tasks.spawn(async move { let mut admitted = Some(admission.replica); - let _retained_reservations = reservations; + let mut retained_reservations = reservations; let mut publication_proofs = Some(proofs); // Admission delay cannot restart or extend the existing retry grace. let fleet_deadline = admission.fleet_deadline; @@ -539,6 +539,64 @@ pub(super) fn start_admitted_publication( let mut authority = std::time::Duration::ZERO; let result = async { let preparation_started = std::time::Instant::now(); + if let Some(captures) = super::bundle::selected_prefix( + &publisher, + &durabilities, + published_commit_sequence, + merged.position, + ) + .await? + { + let selected = Arc::clone( + captures + .last() + .ok_or(Error::Control("bundle publication lacks capture"))? + .selected(), + ); + // No upload can still be reading these files: this task owns + // the sole publisher token and has not begun root preparation. + // Drop the old admission before fresh origin reconstruction. + drop(admitted.take()); + pool.release_bundle_captures(cell, captures).await?; + drop(merged); + for (mut pending, reservation) in + pendings.drain(..).zip(retained_reservations.iter_mut()) + { + pending.release_selected_metadata(); + let bytes = usize::try_from(pending.retained_memory_bytes()) + .map_err(|_| Error::Capacity("bundle outcome memory"))?; + drop(pending); + reservation.shrink_retained(bytes)?; + } + preparation = preparation_started.elapsed(); + let authority_started = std::time::Instant::now(); + let root = loop { + match publisher + .materialize_bundle_with_due(&selected.proof, published_next_due_ms) + .await + { + Ok(root) => break root, + Err(error) if is_storage_publication_error(&error) => { + if std::time::Instant::now() >= fleet_deadline { + return Err(error); + } + tokio::time::sleep(retry_delay).await; + retry_delay = retry_delay + .saturating_mul(2) + .min(std::time::Duration::from_secs(2)); + } + Err(error) => return Err(error), + } + }; + pool.bind_bundle_materialized(cell, root).await?; + PendingDurability::prove_objects(&durabilities).await?; + let published = pool.confirm_published_range(cell, root).await?; + authority = authority_started.elapsed(); + if published != expected { + return Err(Error::Control("bundle result differs from queued commit")); + } + return Ok(()); + } let prepared = loop { let attempt = if let Some(replica) = admitted.take() { publisher diff --git a/crates/cellule-runtime/src/cell/actor/runtime.rs b/crates/cellule-runtime/src/cell/actor/runtime.rs index bd1eee32..5682da92 100644 --- a/crates/cellule-runtime/src/cell/actor/runtime.rs +++ b/crates/cellule-runtime/src/cell/actor/runtime.rs @@ -314,6 +314,7 @@ impl CellRuntime { "Cell runtime node durability was initialized twice", )); } + durability.attach_selection_resources(self.inner.pool.resource_ledger())?; *slot = Some((application, durability)); Ok(()) } @@ -378,6 +379,7 @@ impl CellRuntime { "Cell runtime node durability epoch did not advance", )); } + durability.attach_selection_resources(self.inner.pool.resource_ledger())?; let (_, previous) = slot .replace((application, durability)) .ok_or(Error::Control("Cell runtime node durability disappeared"))?; diff --git a/crates/cellule-runtime/src/cell/executor/bundle.rs b/crates/cellule-runtime/src/cell/executor/bundle.rs new file mode 100644 index 00000000..7a1f30e6 --- /dev/null +++ b/crates/cellule-runtime/src/cell/executor/bundle.rs @@ -0,0 +1,88 @@ +//! Exact selected-capture cleanup before independently joined root materialization. + +use super::*; +use crate::publication::VerifiedBundleCapture; + +impl CellExecutor { + pub(crate) fn release_bundle_captures( + &mut self, + captures: &[VerifiedBundleCapture], + ) -> Result<()> { + if self.fenced || self.pending_migration.is_some() || self.bundle_materialization.is_some() + { + return Err(Error::PendingPublication); + } + let newest = captures + .last() + .ok_or(Error::Control("bundle cleanup lacks captures"))?; + let pending = self + .pending + .get(captures.len() - 1) + .ok_or(Error::PendingPublication)?; + let selected = newest.selected(); + if selected.proof.commit_sequence() != pending.outcome.commit_sequence() + || selected.proof.position() != pending.cuts.position + { + return Err(Error::Control("bundle cleanup endpoint differs")); + } + // Validate the whole oldest prefix before deleting any file. A selected + // suffix, endpoint match alone, or another original assignment cannot + // retire an earlier unresolved capture. + for (pending, capture) in self.pending.iter().zip(captures) { + if pending.prepared.is_some() + || capture.selected().proof.binding() != selected.proof.binding() + { + return Err(Error::Control( + "bundle cleanup changes publication ownership", + )); + } + capture.verify_pending(self.cell, self.incarnation, pending)?; + } + for pending in self.pending.iter().take(captures.len()) { + self.db.prune_captured(&pending.cuts)?; + } + // Keep outcomes and coordination pending until the actor's sole + // publisher materializes the exact endpoint. Proven visibility may + // advance now; an unproven later suffix still blocks queries/retries. + for pending in self.pending.iter_mut().take(captures.len()) { + self.pending_bytes = self + .pending_bytes + .checked_sub(pending.retained_bytes()) + .ok_or(Error::Control("bundle cleanup accounting underflow"))?; + pending.cuts.segments = Vec::new(); + pending.durable = true; + } + self.bundle_materialization = Some(std::sync::Arc::clone(selected)); + Ok(()) + } + + pub(crate) fn bind_bundle_materialized(&mut self, root: &cellule_ltx::RootRef) -> Result<()> { + let selected = self + .bundle_materialization + .as_ref() + .ok_or(Error::PendingPublication)?; + if root.cell != *self.cell.as_bytes() + || root.incarnation != *self.incarnation.as_bytes() + || root.commit_sequence != selected.proof.commit_sequence() + || root.position != selected.proof.position() + { + return Err(Error::Control( + "materialized root differs from bundle cleanup", + )); + } + let index = self + .pending + .iter() + .position(|pending| pending.outcome.commit_sequence() == root.commit_sequence) + .ok_or(Error::PendingPublication)?; + if self.pending.iter().take(index + 1).any(|pending| { + !pending.durable || !pending.cuts.segments.is_empty() || pending.prepared.is_some() + }) { + return Err(Error::Control("materialized bundle omits retained capture")); + } + for pending in self.pending.iter_mut().take(index + 1) { + pending.prepared = Some(*root); + } + Ok(()) + } +} diff --git a/crates/cellule-runtime/src/cell/executor/mod.rs b/crates/cellule-runtime/src/cell/executor/mod.rs index 3936a73a..c0ac57db 100644 --- a/crates/cellule-runtime/src/cell/executor/mod.rs +++ b/crates/cellule-runtime/src/cell/executor/mod.rs @@ -9,6 +9,7 @@ use crate::identity::{IncarnationId, RequestId}; use crate::primitives::maintenance::TransferWorkInventory; use crate::{Error, Result}; +mod bundle; mod group; pub(crate) use group::{MAX_NATIVE_GROUP, NativeCommand, NativeGroupExecution}; @@ -214,6 +215,10 @@ pub struct MigrationOutcome { } impl PendingCommit { + pub(crate) fn release_selected_metadata(&mut self) { + self.cuts.segments = Vec::new(); + } + /// Returns the durable outcome awaiting publication. #[must_use] pub fn outcome(&self) -> &StoredOutcome { @@ -281,8 +286,9 @@ pub enum CommandExecution { /// /// The actor may continue after the preceding commit has a durability proof. /// Dropping a caller does not remove queued cuts or their result. Only -/// `confirm_published` releases retained files after an authoritative root -/// matches the oldest local commit. +/// an exact original bundle receipt or `confirm_published` can release retained +/// files. Outcomes remain pending until the authoritative root matches the +/// covered local commits. pub struct CellExecutor { db: Db, cell: CellId, @@ -291,6 +297,7 @@ pub struct CellExecutor { pending: VecDeque, pending_bytes: u64, published_sequence: u64, + bundle_materialization: Option>, pending_migration: Option, fenced: bool, } @@ -343,6 +350,7 @@ impl CellExecutor { pending: VecDeque::new(), pending_bytes: 0, published_sequence: 0, + bundle_materialization: None, pending_migration: None, fenced: false, } @@ -1154,6 +1162,16 @@ impl CellExecutor { outcomes.push(pending.outcome); } self.published_sequence = root.commit_sequence; + if self + .bundle_materialization + .as_ref() + .is_some_and(|selected| { + selected.proof.commit_sequence() == root.commit_sequence + && selected.proof.position() == root.position + }) + { + self.bundle_materialization = None; + } Ok(outcomes) } diff --git a/crates/cellule-runtime/src/cell/worker/mod.rs b/crates/cellule-runtime/src/cell/worker/mod.rs index 132c04be..ce4dd17d 100644 --- a/crates/cellule-runtime/src/cell/worker/mod.rs +++ b/crates/cellule-runtime/src/cell/worker/mod.rs @@ -793,6 +793,38 @@ impl SqlWorkerPool { receive(response).await } + pub(crate) async fn release_bundle_captures( + &self, + cell: CellId, + captures: Vec, + ) -> Result<()> { + let (reply, response) = oneshot::channel(); + self.send( + cell, + WorkerCommand::ReleaseBundleCaptures { + cell, + captures, + reply, + }, + ) + .await?; + receive(response).await + } + + pub(crate) async fn bind_bundle_materialized( + &self, + cell: CellId, + root: cellule_ltx::RootRef, + ) -> Result<()> { + let (reply, response) = oneshot::channel(); + self.send( + cell, + WorkerCommand::BindBundleMaterialized { cell, root, reply }, + ) + .await?; + receive(response).await + } + pub(crate) async fn confirm_migration_published( &self, cell: CellId, @@ -1267,6 +1299,16 @@ enum WorkerCommand { commit_sequence: u64, reply: oneshot::Sender>, }, + ReleaseBundleCaptures { + cell: CellId, + captures: Vec, + reply: oneshot::Sender>, + }, + BindBundleMaterialized { + cell: CellId, + root: cellule_ltx::RootRef, + reply: oneshot::Sender>, + }, ConfirmBootstrapPublished { cell: CellId, cuts: Box, diff --git a/crates/cellule-runtime/src/cell/worker/run.rs b/crates/cellule-runtime/src/cell/worker/run.rs index 912759bb..8d786cfb 100644 --- a/crates/cellule-runtime/src/cell/worker/run.rs +++ b/crates/cellule-runtime/src/cell/worker/run.rs @@ -401,6 +401,24 @@ fn run_worker_command( .and_then(|cell| cell.executor.confirm_durable(commit_sequence)); let _ = reply.send(result); } + WorkerCommand::ReleaseBundleCaptures { + cell, + captures, + reply, + } => { + let result = cells + .get_mut(&cell) + .ok_or(Error::CellNotActive) + .and_then(|cell| cell.executor.release_bundle_captures(&captures)); + let _ = reply.send(result); + } + WorkerCommand::BindBundleMaterialized { cell, root, reply } => { + let result = cells + .get_mut(&cell) + .ok_or(Error::CellNotActive) + .and_then(|cell| cell.executor.bind_bundle_materialized(&root)); + let _ = reply.send(result); + } WorkerCommand::ConfirmBootstrapPublished { cell, cuts, reply } => { let result = cells .get_mut(&cell) diff --git a/crates/cellule-runtime/src/fleet/resource.rs b/crates/cellule-runtime/src/fleet/resource.rs index 5372f339..dc993cc3 100644 --- a/crates/cellule-runtime/src/fleet/resource.rs +++ b/crates/cellule-runtime/src/fleet/resource.rs @@ -330,6 +330,10 @@ pub(crate) struct LedgerState { } impl ResourceLedger { + pub(crate) fn same_ledger(&self, other: &Self) -> bool { + Arc::ptr_eq(&self.state, &other.state) + } + pub(crate) fn new(limit: ResourceCost) -> Self { Self { state: Arc::new(LedgerState { @@ -486,6 +490,22 @@ pub(crate) struct ResourceReservation { cost: ResourceCost, } +impl ResourceReservation { + /// Returns only memory whose owned capture indexes have already been dropped. + pub(crate) fn shrink_retained(&mut self, bytes: usize) -> Result<()> { + let released = self + .cost + .retained_bytes + .checked_sub(bytes) + .ok_or(Error::Capacity("capture cleanup cannot grow admission"))?; + self.cost.retained_bytes = bytes; + self.ledger + .release(ResourceCost::zero().with_retained_bytes(released)); + self.ledger.state.released.notify_waiters(); + Ok(()) + } +} + impl Drop for ResourceReservation { fn drop(&mut self) { self.ledger.release(self.cost); @@ -606,6 +626,22 @@ impl cellule_ltx::HostResourceAdmission for LedgerHostResourceAdmission { #[cfg(test)] mod tests { use super::*; + + #[test] + fn capture_cleanup_returns_only_released_memory_and_preserves_outcome_admission() { + let cost = ResourceCost::active_cell().with_retained_bytes(128); + let ledger = ResourceLedger::new(cost); + let mut held = ledger.try_reserve(cost).unwrap(); + held.shrink_retained(32).unwrap(); + assert_eq!( + ledger.snapshot().unwrap().used, + ResourceCost::active_cell().with_retained_bytes(32) + ); + assert!(held.shrink_retained(64).is_err()); + assert_eq!(ledger.snapshot().unwrap().used.retained_bytes(), 32); + drop(held); + assert_eq!(ledger.snapshot().unwrap().used, ResourceCost::zero()); + } use cellule_ltx::HostResourceAdmission; #[tokio::test] diff --git a/crates/cellule-runtime/src/fleet/telemetry.rs b/crates/cellule-runtime/src/fleet/telemetry.rs index a19eb2fc..126ab7f1 100644 --- a/crates/cellule-runtime/src/fleet/telemetry.rs +++ b/crates/cellule-runtime/src/fleet/telemetry.rs @@ -14,6 +14,8 @@ pub enum CommandResponseSource { Fleet, /// Object publication proved the new commit. Object, + /// Original shared bundle selection proved the complete captured commit. + Bundle, } /// One completed object publication, which may finish after a follower-proof diff --git a/crates/cellule-runtime/src/node/bundle/mod.rs b/crates/cellule-runtime/src/node/bundle/mod.rs index 26939212..96332663 100644 --- a/crates/cellule-runtime/src/node/bundle/mod.rs +++ b/crates/cellule-runtime/src/node/bundle/mod.rs @@ -159,6 +159,16 @@ pub struct BundleCoverageProof { live: Option, } impl BundleCoverageProof { + pub(crate) fn check_live_assignment( + &self, + assignment: &crate::node::log::AssignedCommitRange, + ) -> Result<()> { + self.live + .as_ref() + .ok_or(Error::Node("cold bundle cannot release live capture"))? + .check_assignment(assignment) + } + /// Original Cell authority pin. pub fn binding(&self) -> BundleBindingRef { self.pin @@ -182,6 +192,37 @@ impl BundleCoverageProof { .ltx_root() .ok_or(Error::Node("bundle binding has no base")) } + + pub(crate) fn contains_assignment( + &self, + assignment: &crate::node::log::AssignedCommitRange, + ) -> bool { + self.live + .as_ref() + .is_some_and(|live| live.contains_assignment(assignment)) + } + + pub(crate) fn assignment_count(&self) -> usize { + self.live.as_ref().map_or(0, |live| live.assignment_count()) + } + + pub(crate) fn retained_metadata_bytes(&self) -> Result { + let control = self.binding.control.encode()?.len(); + self.binding + .locators + .capacity() + .checked_mul(std::mem::size_of::()) + .and_then(|bytes| { + bytes.checked_add( + self.live + .as_ref() + .map_or(0, |live| live.retained_metadata_bytes()), + ) + }) + .and_then(|bytes| bytes.checked_add(control.checked_mul(4)?)) + .and_then(|bytes| bytes.checked_add(std::mem::size_of::() + 256)) + .ok_or(Error::Capacity("selected bundle metadata")) + } } impl Catalog { diff --git a/crates/cellule-runtime/src/node/bundle/selection.rs b/crates/cellule-runtime/src/node/bundle/selection.rs index cbc6526f..728d70f9 100644 --- a/crates/cellule-runtime/src/node/bundle/selection.rs +++ b/crates/cellule-runtime/src/node/bundle/selection.rs @@ -9,6 +9,34 @@ pub(super) struct LiveBundleCoverage { assignments: Vec, } +impl LiveBundleCoverage { + pub(super) fn check_assignment( + &self, + assignment: &crate::node::log::AssignedCommitRange, + ) -> Result<()> { + self.lease.check()?; + if !self.contains_assignment(assignment) { + return Err(Error::Node("bundle lacks original assigned capture")); + } + Ok(()) + } + + pub(super) fn assignment_count(&self) -> usize { + self.assignments.len() + } + + pub(super) fn contains_assignment( + &self, + assignment: &crate::node::log::AssignedCommitRange, + ) -> bool { + self.assignments.contains(assignment) + } + + pub(super) fn retained_metadata_bytes(&self) -> usize { + self.assignments.capacity() * std::mem::size_of::() + } +} + impl NodeDirectory { /// Verifies a contiguous complete native range and uploads one proposal for /// every participating Cell. Neither upload nor this value grants an ACK. diff --git a/crates/cellule-runtime/src/node/bundle/tests/actor.rs b/crates/cellule-runtime/src/node/bundle/tests/actor.rs new file mode 100644 index 00000000..3d783b8d --- /dev/null +++ b/crates/cellule-runtime/src/node/bundle/tests/actor.rs @@ -0,0 +1,603 @@ +//! The real actor ACK/read/retry path, with original authority and durable peers. +use super::*; +use crate::cell::actor::CellRuntime; +use crate::cell::catalog::{CatalogEntry, CatalogRole, CellCatalog}; +use crate::cell::executor::{HandlerOutcome, MutationIdentity, Resolution}; +use crate::cell::worker::SqlWorkerPool; +use crate::fleet::telemetry::{CellTelemetry, CommandResponseSource}; +use crate::identity::{CellTarget, NamespaceId, RequestId, TenantId}; +use crate::node::durability::{NodeBundleAuthority, NodeDurability, NodeLogAuthority}; +use crate::node::log_shipper::NodeLogShipper; +use crate::node::log_transport::{ + AppendRequest, LocalFollowerTransport, NodeLogTransport, RetireRequest, SealRequest, + TailRequest, +}; +use futures_util::future::BoxFuture; +use std::sync::{ + Mutex, + atomic::{AtomicBool, Ordering}, +}; + +pub(super) struct Authority { + pub(super) directory: NodeDirectory, + pub(super) observed: tokio::sync::Mutex, +} + +impl NodeBundleAuthority for Authority { + fn bind<'a>( + &'a self, + authority: &'a CellAuthority, + observed: &'a VersionedControl, + ) -> BoxFuture<'a, Result> { + Box::pin(async move { + let mut node = self.observed.lock().await; + let (next, pinned) = self + .directory + .bind_bundle_cell(&node, authority, observed, NOW) + .await?; + *node = next; + Ok(pinned) + }) + } + fn close<'a>( + &'a self, + authority: &'a CellAuthority, + observed: &'a VersionedControl, + issued: crate::node::log::CellIssuedRange, + ) -> BoxFuture<'a, Result<()>> { + Box::pin(async move { + let mut node = self.observed.lock().await; + let proof = self + .directory + .load_bundle_coverage(authority, observed, Limits::default()) + .await?; + *node = self + .directory + .checkpoint_bundle_cell(&node, authority, &proof, Limits::default(), NOW) + .await?; + *node = self + .directory + .begin_bundle_close(&node, proof.binding(), issued, NOW) + .await?; + *node = self + .directory + .finish_bundle_close(&node, proof.binding(), issued, NOW) + .await?; + Ok(()) + }) + } +} + +impl NodeLogAuthority for Authority { + fn activate<'a>(&'a self, _: u64) -> BoxFuture<'a, Result<()>> { + Box::pin(async move { + let mut node = self.observed.lock().await; + *node = self.directory.activate_log(&node, NOW).await?; + Ok(()) + }) + } + fn advance_coverage<'a>(&'a self, _: u64, through: u64) -> BoxFuture<'a, Result<()>> { + Box::pin(async move { + let mut node = self.observed.lock().await; + *node = self + .directory + .advance_log_coverage(&node, through, NOW) + .await?; + Ok(()) + }) + } + fn close<'a>( + &'a self, + retirement: &'a crate::node::log::NodeLogRetirementObservation, + ) -> BoxFuture<'a, Result<()>> { + Box::pin(async move { + retirement.confirmed()?; + let mut node = self.observed.lock().await; + *node = self + .directory + .close_log(&node, retirement.barrier(), NOW) + .await?; + Ok(()) + }) + } +} + +pub(super) struct PausedFollowers { + peers: [LocalFollowerTransport; 2], + released: AtomicBool, + changed: tokio::sync::Notify, +} + +pub(super) fn transport(f: &Fixture, released: bool) -> Arc { + let peers = [2, 3].map(|byte| { + let store = crate::follower::FollowerStore::open( + f.scratch.path().join(format!("follower-{byte}")), + Limits::default(), + cellule_ltx::DiskBudget::new(32 << 20), + ) + .unwrap(); + LocalFollowerTransport::new(NodeId::from_bytes([byte; 16]), store) + }); + Arc::new(PausedFollowers { + peers, + released: AtomicBool::new(released), + changed: tokio::sync::Notify::new(), + }) +} + +impl NodeLogTransport for PausedFollowers { + fn append<'a>( + &'a self, + member: NodeId, + request: AppendRequest, + ) -> BoxFuture<'a, Result> { + Box::pin(async move { + while !self.released.load(Ordering::Acquire) { + let changed = self.changed.notified(); + tokio::pin!(changed); + changed.as_mut().enable(); + if !self.released.load(Ordering::Acquire) { + changed.await; + } + } + let index = usize::from(member == NodeId::from_bytes([3; 16])); + self.peers[index].append(member, request).await + }) + } + fn seal<'a>( + &'a self, + member: NodeId, + request: SealRequest, + ) -> BoxFuture<'a, Result> { + self.peers[usize::from(member == NodeId::from_bytes([3; 16]))].seal(member, request) + } + fn retire<'a>( + &'a self, + member: NodeId, + request: RetireRequest, + ) -> BoxFuture<'a, Result> { + self.peers[usize::from(member == NodeId::from_bytes([3; 16]))].retire(member, request) + } + fn tail<'a>( + &'a self, + member: NodeId, + request: TailRequest, + ) -> BoxFuture<'a, Result>> { + self.peers[usize::from(member == NodeId::from_bytes([3; 16]))].tail(member, request) + } +} + +#[derive(Default)] +struct Responses(Mutex>); +impl CellTelemetry for Responses { + fn command_response( + &self, + source: CommandResponseSource, + _: std::time::Duration, + _: std::time::Duration, + ) { + self.0.lock().unwrap().push(source); + } +} + +#[tokio::test(flavor = "multi_thread")] +async fn selected_complete_capture_acks_and_reads_before_root_cas_then_joins_drain_and_restores() { + actor_case(false, false, false).await; +} + +#[tokio::test(flavor = "multi_thread")] +async fn prior_fleet_ack_retains_complete_issued_binding_until_joined_root_and_catalog_closure() { + actor_case(true, false, false).await; +} + +#[tokio::test(flavor = "multi_thread")] +async fn cancelled_bundle_caller_retains_mutation_and_retry_until_joined_drain() { + actor_case(false, true, false).await; +} + +#[tokio::test(flavor = "multi_thread")] +async fn selected_capture_releases_files_before_root_cas_and_joins_origin_materialization() { + actor_case(false, false, true).await; +} + +async fn actor_case(prior_fleet: bool, cancel_caller: bool, early_selection: bool) { + let backend = Arc::new(super::faults::ReplyFault::default()); + let mut f = Fixture::with_store(backend.clone()).await; + super::coverage::enroll(&mut f).await; + let authority = Arc::new(Authority { + directory: f.directory.clone(), + observed: tokio::sync::Mutex::new(f.node.clone()), + }); + let transport = transport(&f, false); + let shipper = + NodeLogShipper::new(f.gate.clone(), transport.clone(), Limits::default()).unwrap(); + let durability = Arc::new(NodeDurability::new( + f.gate.clone(), + shipper, + authority.clone(), + transport.clone(), + f.lease.clone(), + )); + let mut feed = durability + .enable_bundle_publication(authority.clone()) + .unwrap(); + let pool = SqlWorkerPool::new(2, 4).unwrap(); + let dirty = Arc::new(tokio::sync::Semaphore::new(1)); + let runtime = CellRuntime::new_with_replica_host_requiring_node_lease( + pool.clone(), + 32 << 20, + SessionId::from_bytes([1; 16]), + cellule_ltx::Host::default().with_dirty_slots(dirty.clone()), + ) + .unwrap(); + runtime.install_node_lease(f.lease.clone()).unwrap(); + runtime + .install_node_durability(ApplicationId::from_bytes([9; 16]), durability.clone()) + .unwrap(); + let responses = Arc::new(Responses::default()); + runtime.install_telemetry(responses.clone()).unwrap(); + let target = CellTarget::new( + TenantId::from_bytes([1; 16]), + ApplicationId::from_bytes([9; 16]), + NamespaceId::from_bytes([13; 16]), + b"bundle-actor", + ) + .unwrap(); + let catalog = CellCatalog::new(f.layout.clone(), target.tenant()); + let proof = catalog + .provision( + CatalogEntry::new( + &target, + CatalogRole::Application, + Digest::from_bytes([12; 32]), + 1, + ) + .unwrap(), + ) + .await + .unwrap(); + let cell_authority = CellAuthority::new(f.layout.clone()); + let control = cell_authority + .create_initial( + &proof, + IncarnationId::from_bytes([4; 16]), + Owner { + session: SessionId::from_bytes([1; 16]), + endpoint: "https://bundle.internal:8081".into(), + }, + ) + .await + .unwrap(); + let replica = CellReplica::new( + f.layout.clone(), + *target.cell_id().as_bytes(), + [4; 16], + Limits::default(), + ) + .unwrap(); + let handle = runtime + .bootstrap( + proof, + replica.clone(), + cell_authority.clone(), + control, + f.scratch.path().join("actor.sqlite"), + |tx| { + tx.execute_batch( + "CREATE TABLE counter(value INTEGER); INSERT INTO counter VALUES(0)", + )?; + Ok(()) + }, + ) + .await + .unwrap(); + runtime.active_catalog_entries().await.unwrap(); + let before = cell_authority + .load(target.cell_id()) + .await + .unwrap() + .unwrap(); + let root = before.value().ltx_root().unwrap(); + assert!(before.value().bundle_binding.is_some()); + let preparation_hold = if early_selection { + Some(dirty.clone().acquire_owned().await.unwrap()) + } else { + None + }; + backend.mode.store(9, Ordering::SeqCst); + let identity = MutationIdentity { + request_id: RequestId::from_bytes([8; 16]), + issued_at_ms: 10, + expires_at_ms: 10_000, + }; + let digest = Digest::from_bytes([9; 32]); + let client = handle.clone(); + let command = tokio::spawn(async move { + client + .execute(identity, digest, 20, 1024, 1024, |tx| { + tx.execute("UPDATE counter SET value=value+1", [])?; + Ok(HandlerOutcome::Success(b"selected".to_vec())) + }) + .await + }); + let mut command = Some(command); + let capture = tokio::time::timeout(std::time::Duration::from_secs(5), feed.recv()) + .await + .unwrap() + .unwrap(); + if !early_selection { + tokio::time::timeout( + std::time::Duration::from_secs(5), + backend.pin_started.notified(), + ) + .await + .unwrap(); + } + let captured_paths = pool + .pending(target.cell_id()) + .await + .unwrap() + .unwrap() + .cuts() + .segments + .iter() + .map(|segment| segment.path().to_owned()) + .collect::>(); + assert!(!captured_paths.is_empty()); + assert!(!command.as_ref().unwrap().is_finished()); + let prior_outcome = if prior_fleet { + transport.released.store(true, Ordering::Release); + transport.changed.notify_waiters(); + Some( + tokio::time::timeout(std::time::Duration::from_secs(5), command.take().unwrap()) + .await + .unwrap() + .unwrap() + .unwrap(), + ) + } else { + if cancel_caller { + command.as_ref().unwrap().abort(); + } + None + }; + let publication = { + let mut node = authority.observed.lock().await; + let prepared = authority + .directory + .prepare_node_bundle(&node, capture.frames(), &[capture.assignment()], NOW) + .await + .unwrap(); + let (next, proofs) = authority + .directory + .select_node_bundle(&node, &prepared, &f.lease, Limits::default(), NOW) + .await + .unwrap(); + *node = next; + durability + .confirm_selected_captures(std::slice::from_ref(&capture), proofs) + .unwrap() + }; + drop(preparation_hold); + if early_selection { + tokio::time::timeout( + std::time::Duration::from_secs(5), + backend.pin_started.notified(), + ) + .await + .unwrap(); + assert!( + pool.pending(target.cell_id()) + .await + .unwrap() + .unwrap() + .cuts() + .segments + .is_empty() + ); + assert!( + captured_paths.iter().all(|path| !path.exists()), + "exact selected capture must release disk before root CAS" + ); + } + let outcome = if let Some(outcome) = prior_outcome { + outcome + } else if cancel_caller { + assert!(command.take().unwrap().await.unwrap_err().is_cancelled()); + handle + .execute(identity, digest, 21, 1024, 1024, |_| { + panic!("cancelled retry must not execute again") + }) + .await + .unwrap() + } else { + tokio::time::timeout(std::time::Duration::from_secs(5), command.take().unwrap()) + .await + .unwrap() + .unwrap() + .unwrap() + }; + assert_eq!(outcome.commit_sequence(), 1); + let mut suffix_publication = None; + let mut suffix_capture = None; + if early_selection { + let second_handle = handle.clone(); + let second = tokio::spawn(async move { + second_handle + .execute( + MutationIdentity { + request_id: RequestId::from_bytes([9; 16]), + ..identity + }, + digest, + 22, + 1024, + 1024, + |tx| { + tx.execute("UPDATE counter SET value=2", [])?; + Ok(HandlerOutcome::Success(b"later".to_vec())) + }, + ) + .await + }); + let capture = tokio::time::timeout(std::time::Duration::from_secs(5), feed.recv()) + .await + .unwrap() + .unwrap(); + assert!(!second.is_finished()); + let unproven = pool + .query( + target.cell_id(), + 1024, + crate::cell::worker::SqlDeadline::new( + std::time::Instant::now() + std::time::Duration::from_secs(5), + ), + Box::new(|_| panic!("unproven suffix must not reach a query handler")), + ) + .await; + assert!(matches!(unproven, Err(Error::PendingPublication))); + let selected = { + let mut node = authority.observed.lock().await; + let prepared = authority + .directory + .prepare_node_bundle(&node, capture.frames(), &[capture.assignment()], NOW) + .await + .unwrap(); + let (next, proofs) = authority + .directory + .select_node_bundle(&node, &prepared, &f.lease, Limits::default(), NOW) + .await + .unwrap(); + *node = next; + durability + .confirm_selected_captures(std::slice::from_ref(&capture), proofs) + .unwrap() + }; + assert_eq!( + tokio::time::timeout(std::time::Duration::from_secs(5), second) + .await + .unwrap() + .unwrap() + .unwrap() + .commit_sequence(), + 2 + ); + suffix_publication = Some(selected); + suffix_capture = Some(capture); + } + runtime.active_catalog_entries().await.unwrap(); + let expected_responses = if early_selection { + vec![CommandResponseSource::Bundle, CommandResponseSource::Bundle] + } else { + vec![if prior_fleet { + CommandResponseSource::Fleet + } else if cancel_caller { + CommandResponseSource::Recorded + } else { + CommandResponseSource::Bundle + }] + }; + assert_eq!( + responses.0.lock().unwrap().as_slice(), + expected_responses.as_slice() + ); + assert_eq!( + cell_authority + .load(target.cell_id()) + .await + .unwrap() + .unwrap() + .value() + .ltx_root(), + Some(root) + ); + assert_eq!( + handle + .query(1024, 1024, |connection| Ok(connection + .query_row("SELECT value FROM counter", [], |row| row.get::<_, i64>(0))? + .to_le_bytes() + .to_vec())) + .await + .unwrap(), + (if early_selection { 2_i64 } else { 1_i64 }).to_le_bytes() + ); + assert_eq!( + handle.resolve(identity, digest, 21, 1024).await.unwrap(), + Resolution::Committed(outcome.clone()) + ); + assert_eq!( + handle + .execute(identity, digest, 21, 1024, 1024, |_| panic!( + "retry must use the selected outcome" + )) + .await + .unwrap(), + outcome + ); + let drain = handle.drain(); + tokio::pin!(drain); + let early = tokio::time::timeout(std::time::Duration::from_millis(50), &mut drain).await; + backend.pin_resume.notify_one(); + transport.released.store(true, Ordering::Release); + transport.changed.notify_waiters(); + assert!( + early.is_err(), + "selected ACK cannot discharge root/drain obligations" + ); + tokio::time::timeout(std::time::Duration::from_secs(5), &mut drain) + .await + .unwrap() + .unwrap(); + let control = cell_authority + .load(target.cell_id()) + .await + .unwrap() + .unwrap(); + assert!(control.value().bundle_binding.is_none()); + assert_eq!(control.value().state, ControlState::Idle); + let selected_root = control.value().ltx_root().unwrap(); + assert_eq!( + selected_root.commit_sequence, + if early_selection { 2 } else { 1 } + ); + let restored = f.scratch.path().join("cold.sqlite"); + replica + .open_root(&selected_root) + .await + .unwrap() + .restore(&restored) + .await + .unwrap(); + let connection = cellule_ltx::rusqlite::Connection::open(restored).unwrap(); + assert_eq!( + connection + .query_row("SELECT value FROM counter", [], |row| row.get::<_, i64>(0)) + .unwrap(), + if early_selection { 2 } else { 1 } + ); + assert_eq!( + connection + .query_row( + "SELECT result FROM sys_requests WHERE request_id=?1", + [identity.request_id.as_bytes().as_slice()], + |row| row.get::<_, Vec>(0) + ) + .unwrap(), + b"selected" + ); + drop(connection); + drop(publication); + drop(capture); + drop(suffix_publication); + drop(suffix_capture); + runtime.shutdown().await.unwrap(); + assert!(feed.recv().await.is_none()); + assert_eq!( + pool.resource_ledger() + .snapshot() + .unwrap() + .used + .retained_bytes(), + 0 + ); +} diff --git a/crates/cellule-runtime/src/node/bundle/tests/mod.rs b/crates/cellule-runtime/src/node/bundle/tests/mod.rs index d86cf2e9..8ede9952 100644 --- a/crates/cellule-runtime/src/node/bundle/tests/mod.rs +++ b/crates/cellule-runtime/src/node/bundle/tests/mod.rs @@ -15,11 +15,13 @@ use std::sync::Arc; const NOW: i64 = 1_000_000; const EPOCH: u64 = 2; +mod actor; mod coverage; mod faults; mod index; mod lifecycle; mod ranges; +mod receipts; mod recovery; struct Fixture { count: Arc, diff --git a/crates/cellule-runtime/src/node/bundle/tests/receipts.rs b/crates/cellule-runtime/src/node/bundle/tests/receipts.rs new file mode 100644 index 00000000..6c63efea --- /dev/null +++ b/crates/cellule-runtime/src/node/bundle/tests/receipts.rs @@ -0,0 +1,209 @@ +use super::*; +use crate::cell::worker::SqlWorkerPool; +use crate::node::durability::NodeDurability; +use crate::node::log_shipper::{NodeLogShipper, NodeLogSubmission}; + +#[tokio::test] +async fn selection_receipts_require_complete_cohort_and_admission_and_share_one_cell_charge() { + let mut f = Fixture::new().await; + super::coverage::enroll(&mut f).await; + let mut cell = f.cell(4).await; + let transport = super::actor::transport(&f, true); + let authority = Arc::new(super::actor::Authority { + directory: f.directory.clone(), + observed: tokio::sync::Mutex::new(f.node.clone()), + }); + let shipper = + NodeLogShipper::new(f.gate.clone(), transport.clone(), Limits::default()).unwrap(); + let durability = NodeDurability::new( + f.gate.clone(), + shipper, + authority.clone(), + transport, + f.lease.clone(), + ); + let mut feed = durability.take_publication_feed().unwrap(); + let pool = SqlWorkerPool::new(1, 1).unwrap(); + pool.configure_retained_capacity(64).unwrap(); + durability + .attach_selection_resources(pool.resource_ledger()) + .unwrap(); + let mut captures = Vec::new(); + let mut submitted = Vec::new(); + let mut cuts_by_commit = Vec::new(); + for commit in [2, 3] { + cell.db + .transaction(|tx| { + tx.execute( + "INSERT INTO outcomes VALUES(?1,?2)", + [format!("request-{commit}"), format!("result-{commit}")], + ) + }) + .unwrap(); + let cuts = cell.db.capture().unwrap(); + cuts_by_commit.push(cuts.clone()); + submitted.push( + durability + .submit_capture( + NodeLogSubmission::new( + ApplicationId::from_bytes([9; 16]), + cell.control.value().cell, + cell.control.value().incarnation, + cell.control.value().epoch, + commit, + &cuts, + ) + .unwrap(), + ) + .await + .unwrap(), + ); + captures.push(feed.recv().await.unwrap()); + } + let frames = captures + .iter() + .flat_map(|capture| capture.frames().iter().cloned()) + .collect::>(); + let assignments = captures + .iter() + .map(|capture| capture.assignment()) + .collect::>(); + let scope = assignments[0].scope(); + assert!(assignments[0].matches_capture(scope.cell, scope.incarnation, 2, &cuts_by_commit[0])); + assert!(!assignments[0].matches_capture(scope.cell, scope.incarnation, 3, &cuts_by_commit[1])); + let mut altered = cuts_by_commit[0].clone(); + let original = &altered.segments[0]; + let mut metadata = original.info().clone(); + metadata.blake3[0] ^= 1; + altered.segments[0] = cellule_ltx::LocalSegment::new(original.path().to_owned(), metadata); + assert!(!assignments[0].matches_capture(scope.cell, scope.incarnation, 2, &altered)); + let prepared = f + .directory + .prepare_node_bundle(&f.node, &frames, &assignments, NOW) + .await + .unwrap(); + let (selected, proofs) = f + .directory + .select_node_bundle(&f.node, &prepared, &f.lease, Limits::default(), NOW) + .await + .unwrap(); + f.node = selected; + assert_eq!(proofs.len(), 1); + assert!(matches!( + durability.confirm_selected_captures(&captures[..1], proofs), + Err(Error::Node("selection omits original captured assignments")) + )); + assert_eq!(f.gate.tiered_through(), 0); + let (_, proofs) = f + .directory + .select_node_bundle(&f.node, &prepared, &f.lease, Limits::default(), NOW) + .await + .unwrap(); + assert!(matches!( + durability.confirm_selected_captures(&captures, proofs), + Err(Error::Capacity(_)) + )); + assert_eq!(f.gate.tiered_through(), 0); + assert_eq!( + pool.resource_ledger() + .snapshot() + .unwrap() + .used + .retained_bytes(), + 0 + ); + let cold = f + .directory + .load_bundle_coverage(&cell.authority, &cell.control, Limits::default()) + .await + .unwrap(); + assert!(matches!( + durability.confirm_selected_captures(&captures, vec![cold]), + Err(Error::Node("selection omits original captured assignments")) + )); + pool.configure_retained_capacity(1 << 20).unwrap(); + let (_, proofs) = f + .directory + .select_node_bundle(&f.node, &prepared, &f.lease, Limits::default(), NOW) + .await + .unwrap(); + let expected = proofs[0].retained_metadata_bytes().unwrap(); + let publication = durability + .confirm_selected_captures(&captures, proofs) + .unwrap(); + assert_eq!(publication.selected_through(), 2); + assert_eq!( + pool.resource_ledger() + .snapshot() + .unwrap() + .used + .retained_bytes(), + expected + ); + let first = submitted[0] + .selection + .as_ref() + .unwrap() + .selected() + .await + .unwrap(); + let second = submitted[1] + .selection + .as_ref() + .unwrap() + .selected() + .await + .unwrap(); + assert!(Arc::ptr_eq(&first, &second)); + assert!(first.proof.contains_assignment(&submitted[0].assignment)); + assert!(second.proof.contains_assignment(&submitted[1].assignment)); + drop(first); + drop(second); + drop(publication); + drop(captures); + drop(submitted); + assert_eq!( + pool.resource_ledger() + .snapshot() + .unwrap() + .used + .retained_bytes(), + 0 + ); + let proof = f + .directory + .load_bundle_coverage(&cell.authority, &cell.control, Limits::default()) + .await + .unwrap(); + let root = f.publisher(&cell).materialize_bundle(&proof).await.unwrap(); + let current = cell + .authority + .load(cell.control.value().cell) + .await + .unwrap() + .unwrap(); + f.node = f + .directory + .checkpoint_bundle_cell(&f.node, &cell.authority, &proof, Limits::default(), NOW) + .await + .unwrap(); + let issued = f + .gate + .close_cell_issuance(Fixture::scope(&cell), root) + .unwrap(); + f.node = f + .directory + .begin_bundle_close(&f.node, proof.binding(), issued, NOW) + .await + .unwrap(); + f.node = f + .directory + .finish_bundle_close(&f.node, proof.binding(), issued, NOW) + .await + .unwrap(); + assert_eq!(current.value().ltx_root(), Some(root)); + *authority.observed.lock().await = f.node; + durability.shutdown().await.unwrap(); + assert!(feed.recv().await.is_none()); + pool.shutdown().await.unwrap(); +} diff --git a/crates/cellule-runtime/src/node/durability/mod.rs b/crates/cellule-runtime/src/node/durability/mod.rs index a336ae2f..d77c6733 100644 --- a/crates/cellule-runtime/src/node/durability/mod.rs +++ b/crates/cellule-runtime/src/node/durability/mod.rs @@ -18,6 +18,28 @@ use crate::{Error, Result}; mod object_coverage; use object_coverage::ObjectCoverage; +/// Original node authority used to enroll and close the Cells of an installed +/// shared publication feed. Implementations serialize these mutations and shared +/// selection with their original heartbeat/enrollment state. The host owns and +/// joins the feed; the runtime still verifies Cell departure against origin. +pub trait NodeBundleAuthority: Send + Sync { + /// Pins the Serving Cell before its SQL admission opens. + fn bind<'a>( + &'a self, + authority: &'a crate::control::authority::CellAuthority, + observed: &'a crate::control::authority::VersionedControl, + ) -> BoxFuture<'a, Result>; + + /// Joins complete issued coverage, materialization/checkpoint and catalog + /// closure after SQL/capture tasks close, including earlier Fleet ACKs. + fn close<'a>( + &'a self, + authority: &'a crate::control::authority::CellAuthority, + observed: &'a crate::control::authority::VersionedControl, + issued: crate::node::log::CellIssuedRange, + ) -> BoxFuture<'a, Result<()>>; +} + /// Authoritative node-session mutations required by follower durability. /// /// Implementations must serialize these mutations with heartbeat refreshes and @@ -167,9 +189,19 @@ pub struct NodeDurability { retirement: std::sync::Mutex>>, retirement_proof: OnceCell>, closed: std::sync::atomic::AtomicBool, + selection_resources: std::sync::OnceLock, + bundle_authority: std::sync::OnceLock>, } impl NodeDurability { + pub(crate) fn check_lease(&self) -> Result<()> { + self.node_lease.check() + } + + pub(crate) async fn wait_fenced(&self) { + self.node_lease.wait_fenced().await; + } + /// Observes the epoch's durability frontiers; this does not issue a proof. pub fn progress(&self) -> Result { self.gate.progress() @@ -205,6 +237,8 @@ impl NodeDurability { retirement: std::sync::Mutex::new(None), retirement_proof: OnceCell::new(), closed: std::sync::atomic::AtomicBool::new(false), + selection_resources: std::sync::OnceLock::new(), + bundle_authority: std::sync::OnceLock::new(), } } @@ -227,15 +261,107 @@ impl NodeDurability { &self, submission: NodeLogSubmission, ) -> Result<(CommitTicket, crate::node::log::AssignedCommitRange)> { + self.submit_capture(submission) + .await + .map(|capture| (capture.assignment.ticket(), capture.assignment)) + } + + pub(crate) async fn submit_capture( + &self, + submission: NodeLogSubmission, + ) -> Result { self.node_lease.check()?; let ticket = tokio::select! { - result = self.shipper.submit_assigned(submission) => result?, + result = self.shipper.submit_capture(submission) => result?, () = self.node_lease.wait_fenced() => return Err(Error::Fenced), }; self.node_lease.check()?; Ok(ticket) } + pub(crate) fn attach_selection_resources( + &self, + resources: crate::fleet::resource::ResourceLedger, + ) -> Result<()> { + let original = self.selection_resources.get_or_init(|| resources.clone()); + if !original.same_ledger(&resources) { + return Err(Error::Node("bundle selection resource ledger changed")); + } + Ok(()) + } + + /// Installs this epoch's shared feed and original binding/closure authority + /// before native issuance or actor activation. The host must keep accepting + /// and joining complete captures until runtime drain finishes. + pub fn enable_bundle_publication( + &self, + authority: Arc, + ) -> Result { + self.node_lease.check()?; + let feed = self.shipper.take_publication_feed()?; + self.bundle_authority + .set(authority) + .map_err(|_| Error::Node("bundle authority already installed"))?; + Ok(feed) + } + + pub(crate) async fn bind_bundle_cell( + &self, + authority: &crate::control::authority::CellAuthority, + observed: &crate::control::authority::VersionedControl, + ) -> Result { + let Some(bundle) = self.bundle_authority.get() else { + return Ok(observed.clone()); + }; + self.node_lease.check()?; + let bound = bundle.bind(authority, observed).await?; + self.node_lease.check()?; + observed + .value() + .validate_transition(bound.value(), crate::control::Transition::BindBundle)?; + let pin = bound + .value() + .bundle_binding + .ok_or(Error::Node("bundle enrollment lacks original Cell pin"))?; + let (session, _, epoch) = self.identity()?; + if pin.session != session || pin.epoch != epoch { + return Err(Error::Fenced); + } + Ok(bound) + } + + pub(crate) async fn close_bundle_cell( + &self, + authority: &crate::control::authority::CellAuthority, + observed: &crate::control::authority::VersionedControl, + ) -> Result<()> { + if observed.value().bundle_binding.is_none() { + return Ok(()); + } + let bundle = self + .bundle_authority + .get() + .ok_or(Error::Node("bound Cell lost original bundle authority"))?; + self.node_lease.check()?; + let scope = crate::node::log::CellLogScope { + application: crate::identity::ApplicationId::from_bytes( + *authority.layout().application_id(), + ), + cell: observed.value().cell, + incarnation: observed.value().incarnation, + cell_epoch: observed.value().epoch, + }; + let issued = self.close_cell_issuance( + scope, + observed + .value() + .ltx_root() + .ok_or(Error::PendingPublication)?, + )?; + bundle.close(authority, observed, issued).await?; + self.node_lease.check() + } + /// Freezes exact Cell issuance after the original SQL and capture tasks join. pub fn close_cell_issuance( &self, @@ -359,6 +485,80 @@ impl NodeDurability { crate::node::bundle::confirm_selected_coverage(&self.gate, &self.node_lease, proofs) } + /// Confirms a complete admitted feed cohort and returns its exact selected + /// metadata to the original commands. The installed runtime's ledger pays + /// for each Cell proof once until all command/materialization consumers join. + /// Missing, duplicate, cold or foreign assignments cannot wake siblings. + /// Captures still require their separate ordinary root cleanup and drain. + pub fn confirm_selected_captures( + &self, + captures: &[crate::node::log_shipper::AssignedCapture], + proofs: Vec, + ) -> Result { + self.node_lease.check()?; + if captures.is_empty() || captures.len() > 64 || proofs.len() > 64 { + return Err(Error::Capacity("selected capture cohort")); + } + let assignments = proofs + .iter() + .map(|proof| proof.assignment_count()) + .sum::(); + if assignments != captures.len() || proofs.iter().any(|proof| proof.assignment_count() == 0) + { + return Err(Error::Node("selection omits original captured assignments")); + } + for (index, capture) in captures.iter().enumerate() { + let assignment = capture.assignment(); + if captures[..index] + .iter() + .any(|before| before.assignment() == assignment) + || proofs + .iter() + .filter(|proof| proof.contains_assignment(&assignment)) + .count() + != 1 + { + return Err(Error::Node( + "selection differs from original captured cohort", + )); + } + } + let resources = self.selection_resources.get().ok_or(Error::Node( + "bundle publication has no installed runtime resource ledger", + ))?; + let memories = proofs + .iter() + .map(|proof| { + resources.try_reserve( + crate::fleet::resource::ResourceCost::zero() + .with_retained_bytes(proof.retained_metadata_bytes()?), + ) + }) + .collect::>>()?; + // Validate all original gates and leases before sending any receipt. + // Receipt waiters also require the gate's confirmed Bundle source. + let through = + crate::node::bundle::confirm_selected_coverage(&self.gate, &self.node_lease, &proofs)?; + let selected = proofs + .into_iter() + .zip(memories) + .map(|(proof, memory)| { + Arc::new(crate::node::log_shipper::SelectedBundle { + proof, + _memory: memory, + }) + }) + .collect::>(); + for capture in captures { + let proof = selected + .iter() + .find(|selected| selected.proof.contains_assignment(&capture.assignment())) + .ok_or(Error::Node("selected capture lost original assignment"))?; + capture.confirm_selection(Arc::clone(proof)); + } + Ok(crate::node::log_shipper::SelectedBundlePublication { through, selected }) + } + /// Returns this binding's exact enrolled log epoch. pub fn log_epoch(&self) -> Result { self.gate.log_epoch() diff --git a/crates/cellule-runtime/src/node/log/mod.rs b/crates/cellule-runtime/src/node/log/mod.rs index 16f96fd0..6428b8a7 100644 --- a/crates/cellule-runtime/src/node/log/mod.rs +++ b/crates/cellule-runtime/src/node/log/mod.rs @@ -62,6 +62,7 @@ pub struct AssignedCommitRange { commit: u64, position: cellule_ltx::Position, digest: [u8; 32], + capture_digest: [u8; 32], } impl AssignedCommitRange { pub(crate) const fn scope(&self) -> CellLogScope { @@ -72,6 +73,23 @@ impl AssignedCommitRange { pub const fn ticket(&self) -> CommitTicket { self.ticket } + + pub(crate) fn matches_capture( + &self, + cell: CellId, + incarnation: IncarnationId, + commit: u64, + cuts: &cellule_ltx::CaptureBatch, + ) -> bool { + self.scope.cell == cell + && self.scope.incarnation == incarnation + && self.commit == commit + && self.position == cuts.position + && cuts.segments.len() as u64 + == self.ticket.last_sequence - self.ticket.first_sequence + 1 + && self.capture_digest + == capture_digest(cuts.segments.iter().map(cellule_ltx::LocalSegment::info)) + } pub(crate) fn verify(&self, frames: &[cellule_ltx::VerifiedNodeFrame]) -> Result<()> { if frames.is_empty() || frames.len() as u64 != self.ticket.last_sequence - self.ticket.first_sequence + 1 @@ -106,6 +124,22 @@ impl AssignedCommitRange { } } +fn capture_digest<'a>(segments: impl Iterator) -> [u8; 32] { + let mut hash = blake3::Hasher::new(); + hash.update(b"cellule-assigned-capture-v1"); + for segment in segments { + hash.update(&segment.min_txid.to_le_bytes()); + hash.update(&segment.max_txid.to_le_bytes()); + hash.update(&segment.page_size.to_le_bytes()); + hash.update(&segment.database_pages.to_le_bytes()); + hash.update(&segment.pre_checksum.to_le_bytes()); + hash.update(&segment.post_checksum.to_le_bytes()); + hash.update(&segment.size_bytes.to_le_bytes()); + hash.update(&segment.blake3); + } + *hash.finalize().as_bytes() +} + impl CellIssuedRange { /// Original lane session. pub const fn leader_session(&self) -> SessionId { @@ -474,6 +508,7 @@ impl DurabilityGate { commit: scope.commit_sequence, position: frame.segment().position(), digest: [0; 32], + capture_digest: [0; 32], }) } Some(assignment) => { @@ -553,6 +588,7 @@ impl DurabilityGate { hash.update(&frame.digest()); } assignment.digest = *hash.finalize().as_bytes(); + assignment.capture_digest = capture_digest(frames.iter().map(|frame| frame.segment())); } Ok(assignment) } diff --git a/crates/cellule-runtime/src/node/log_shipper/mod.rs b/crates/cellule-runtime/src/node/log_shipper/mod.rs index 0a9e978c..93af51dc 100644 --- a/crates/cellule-runtime/src/node/log_shipper/mod.rs +++ b/crates/cellule-runtime/src/node/log_shipper/mod.rs @@ -12,7 +12,8 @@ use crate::node::log_transport::{AppendRequest, NodeLogTransport}; use crate::{Error, Result}; mod publication; -pub use publication::{AssignedCapture, NodePublicationFeed}; +pub use publication::{AssignedCapture, NodePublicationFeed, SelectedBundlePublication}; +pub(crate) use publication::{SelectedBundle, SubmittedCapture}; const MAX_BATCH_FRAMES: usize = 64; const MAX_QUEUED_SUBMISSIONS: usize = 512; @@ -312,6 +313,15 @@ impl NodeLogShipper { &self, submission: NodeLogSubmission, ) -> Result<(CommitTicket, crate::node::log::AssignedCommitRange)> { + self.submit_capture(submission) + .await + .map(|capture| (capture.assignment.ticket(), capture.assignment)) + } + + pub(crate) async fn submit_capture( + &self, + submission: NodeLogSubmission, + ) -> Result { let frame_count = submission.frame_count()?; if submission .segments @@ -365,13 +375,14 @@ impl NodeLogShipper { let reservation = Arc::new(OutstandingBytes { _permit: reservation, }); - if let Some(publication) = publication { - publication.send(AssignedCapture::new( - assignment, - encoded.clone(), - Arc::clone(&reservation), - )); - } + let selection = if let Some(publication) = publication { + let (capture, selection) = + AssignedCapture::new(assignment, encoded.clone(), Arc::clone(&reservation)); + publication.send(capture); + Some(selection) + } else { + None + }; let frames = encoded .into_iter() .enumerate() @@ -382,7 +393,10 @@ impl NodeLogShipper { }) .collect(); slot.send(QueuedSubmission { frames }); - Ok((ticket, assignment)) + Ok(SubmittedCapture { + assignment, + selection, + }) } /// Closes admission and drains every accepted frame to the current epoch. diff --git a/crates/cellule-runtime/src/node/log_shipper/publication/mod.rs b/crates/cellule-runtime/src/node/log_shipper/publication/mod.rs index bb4e6526..a5b57092 100644 --- a/crates/cellule-runtime/src/node/log_shipper/publication/mod.rs +++ b/crates/cellule-runtime/src/node/log_shipper/publication/mod.rs @@ -4,6 +4,56 @@ use super::*; use std::sync::{Mutex, OnceLock}; use tokio::sync::watch; +/// Selected coverage and its node-ledger reservation, shared by complete +/// captures from the same Cell in one selection. No native bodies are retained. +pub(crate) struct SelectedBundle { + pub(crate) proof: crate::node::bundle::BundleCoverageProof, + pub(crate) _memory: crate::fleet::resource::ResourceReservation, +} + +#[derive(Clone)] +pub(crate) struct CaptureSelection { + receiver: watch::Receiver>>, +} + +impl CaptureSelection { + pub(crate) async fn selected(&self) -> Result> { + let mut receiver = self.receiver.clone(); + loop { + if let Some(selected) = receiver.borrow_and_update().clone() { + return Ok(selected); + } + receiver.changed().await.map_err(|_| Error::RuntimeClosed)?; + } + } +} + +#[derive(Clone)] +pub(crate) struct SubmittedCapture { + pub(crate) assignment: crate::node::log::AssignedCommitRange, + pub(crate) selection: Option, +} + +/// Admitted selected metadata retained through command confirmation and root +/// materialization. Every consumer shares the same reservation for a Cell. +pub struct SelectedBundlePublication { + pub(crate) through: u64, + pub(crate) selected: Vec>, +} + +impl SelectedBundlePublication { + /// Contiguous native frontier selected by the original canonical node CAS. + pub const fn selected_through(&self) -> u64 { + self.through + } + + /// Exact Cell proofs; keep this publication alive through joined I/O so its + /// metadata remains charged to the serving node's original resource ledger. + pub fn proofs(&self) -> impl Iterator { + self.selected.iter().map(|selected| &selected.proof) + } +} + /// One complete captured assignment, in the original node-log order. /// /// The native lane constructed and verified these frames before issuance. This @@ -13,6 +63,7 @@ pub struct AssignedCapture { assignment: crate::node::log::AssignedCommitRange, frames: Vec, _reservation: Arc, + selection: watch::Sender>>, } impl AssignedCapture { @@ -20,12 +71,17 @@ impl AssignedCapture { assignment: crate::node::log::AssignedCommitRange, frames: Vec, reservation: Arc, - ) -> Self { - Self { - assignment, - frames, - _reservation: reservation, - } + ) -> (Self, CaptureSelection) { + let (selection, receiver) = watch::channel(None); + ( + Self { + assignment, + frames, + _reservation: reservation, + selection, + }, + CaptureSelection { receiver }, + ) } /// Exact original complete-capture witness accepted by bundle selection. @@ -38,6 +94,12 @@ impl AssignedCapture { pub fn frames(&self) -> &[cellule_ltx::VerifiedNodeFrame] { &self.frames } + + pub(crate) fn confirm_selection(&self, selected: Arc) { + // Selection confirmation is retained even if a cancelled command has + // dropped its receiver. Publication/drain still owns the original cut. + self.selection.send_replace(Some(selected)); + } } /// Sole ordered consumer for this original native epoch's publication work. diff --git a/crates/cellule-runtime/src/publication/mod.rs b/crates/cellule-runtime/src/publication/mod.rs index 55b5f827..105da03b 100644 --- a/crates/cellule-runtime/src/publication/mod.rs +++ b/crates/cellule-runtime/src/publication/mod.rs @@ -29,6 +29,7 @@ pub(crate) struct CellDurabilitySubmitter { cell: crate::CellId, incarnation: crate::identity::IncarnationId, epoch: u64, + binding: Option, node_lease: Option, node_durability: Option, telemetry: crate::fleet::telemetry::CellTelemetryHandle, @@ -106,6 +107,15 @@ impl CellPublisher { pub async fn materialize_bundle( &mut self, proof: &crate::node::bundle::BundleCoverageProof, + ) -> Result { + self.materialize_bundle_with_due(proof, self.observed.value().next_due_ms) + .await + } + + pub(crate) async fn materialize_bundle_with_due( + &mut self, + proof: &crate::node::bundle::BundleCoverageProof, + next_due_ms: Option, ) -> Result { self.check_node_lease()?; if self.observed.value().bundle_binding != Some(proof.binding()) { @@ -127,8 +137,7 @@ impl CellPublisher { self.lineage_confirmed = *confirmation .lock() .map_err(|_| Error::Peer("root lineage confirmation lock poisoned"))?; - self.publish_prepared(&prepared, self.observed.value().next_due_ms) - .await + self.publish_prepared(&prepared, next_due_ms).await } pub(crate) fn with_shared_publication( @@ -161,12 +170,30 @@ impl CellPublisher { .await } + pub(crate) async fn enroll_bundle(&mut self) -> Result<()> { + let Some(slot) = self.node_durability.as_ref() else { + return Ok(()); + }; + let durability = slot + .read() + .map_err(|_| Error::Control("Cell runtime node durability lock poisoned"))? + .as_ref() + .map(|(_, durability)| std::sync::Arc::clone(durability)); + if let Some(durability) = durability { + self.observed = durability + .bind_bundle_cell(&self.authority, &self.observed) + .await?; + } + Ok(()) + } + pub(crate) fn durability_submitter(&self) -> CellDurabilitySubmitter { let control = self.observed.value(); CellDurabilitySubmitter { cell: control.cell, incarnation: control.incarnation, epoch: control.epoch, + binding: control.bundle_binding, node_lease: self.node_lease.clone(), node_durability: self.node_durability.clone(), telemetry: self.telemetry.clone(), @@ -1047,6 +1074,20 @@ impl CellPublisher { /// Releases ownership after the SQL worker has closed the drained Cell. pub(crate) async fn release(&mut self) -> Result<()> { self.check_node_lease()?; + if self.observed.value().bundle_binding.is_some() { + let durability = self + .node_durability + .as_ref() + .ok_or(Error::Node("bound Cell lacks original node durability"))? + .read() + .map_err(|_| Error::Control("Cell runtime node durability lock poisoned"))? + .as_ref() + .map(|(_, durability)| std::sync::Arc::clone(durability)) + .ok_or(Error::Node("bound Cell lacks original node durability"))?; + durability + .close_bundle_cell(&self.authority, &self.observed) + .await?; + } let mut backoff = Backoff::default(); loop { self.check_node_lease()?; @@ -1142,24 +1183,93 @@ impl CellPublisher { } } +struct PendingBundleCapture { + submitted: crate::node::log_shipper::SubmittedCapture, + binding: Option, +} + #[derive(Clone)] pub(crate) struct PendingDurability { durability: std::sync::Arc, ticket: CommitTicket, + capture: Option>, submitted_at: std::time::Instant, telemetry: crate::fleet::telemetry::CellTelemetryHandle, } +/// Original whole-capture receipt. Only the submitting durability handle creates it. +pub(crate) struct VerifiedBundleCapture { + assignment: crate::node::log::AssignedCommitRange, + selected: std::sync::Arc, +} + +impl VerifiedBundleCapture { + pub(crate) fn selected(&self) -> &std::sync::Arc { + &self.selected + } + + pub(crate) fn verify_pending( + &self, + cell: crate::identity::CellId, + incarnation: crate::identity::IncarnationId, + pending: &crate::cell::executor::PendingCommit, + ) -> Result<()> { + self.selected + .proof + .check_live_assignment(&self.assignment)?; + if !self.assignment.matches_capture( + cell, + incarnation, + pending.outcome().commit_sequence(), + pending.cuts(), + ) { + return Err(Error::Control("bundle does not match worker capture")); + } + Ok(()) + } +} + impl PendingDurability { + pub(crate) fn has_bundle_capture(&self) -> bool { + self.capture.is_some() + } + + pub(crate) async fn selected_capture(&self) -> Result> { + let Some(capture) = &self.capture else { + return Ok(None); + }; + let selection = capture.submitted.selection.as_ref().ok_or(Error::Node( + "bundle response lacks original selection receipt", + ))?; + let selected = tokio::select! { + selected = selection.selected() => selected?, + () = self.durability.wait_fenced() => return Err(Error::Fenced), + }; + if capture.submitted.assignment.ticket() != self.ticket + || !selected + .proof + .contains_assignment(&capture.submitted.assignment) + || capture.binding != Some(selected.proof.binding()) + { + return Err(Error::Node( + "bundle response differs from original Cell binding", + )); + } + self.durability.check_lease()?; + Ok(Some(VerifiedBundleCapture { + assignment: capture.submitted.assignment, + selected, + })) + } + pub(crate) async fn prove(&self) -> Result { let proof = self.durability.prove(self.ticket).await?; if proof.source() == crate::node::log::DurabilitySource::Bundle { - // Ordinary actors still confirm visibility through per-Cell roots. - // Their existing object fallback must win until exact bundle proofs - // are connected to command, query, retry and capture release. - return Err(Error::Node( - "shared bundle actor response path is not installed", - )); + self.selected_capture() + .await? + .ok_or(Error::Node("bundle response lacks original capture"))?; + self.telemetry + .durability_proof(proof.source(), self.submitted_at.elapsed()); } if proof.source() == crate::node::log::DurabilitySource::Fleet { self.telemetry @@ -1257,8 +1367,8 @@ impl CellDurabilitySubmitter { commit_sequence, cuts, )?; - let ticket = match durability.submit(submission).await { - Ok(ticket) => ticket, + let capture = match durability.submit_capture(submission).await { + Ok(capture) => capture, Err(error) => { self.check_node_lease()?; // The commit still succeeds through object coverage, so this @@ -1276,9 +1386,20 @@ impl CellDurabilitySubmitter { }; self.telemetry .durability_submission(DurabilitySubmissionOutcome::Fleet); + let ticket = capture.assignment.ticket(); + // Retain the exact assignment only for an installed publication feed. + // Shared ownership avoids copying it for each command-proof waiter; + // the ordinary follower path needs only its existing commit ticket. + let capture = capture.selection.is_some().then(|| { + std::sync::Arc::new(PendingBundleCapture { + submitted: capture, + binding: self.binding, + }) + }); Ok(Some(PendingDurability { durability: std::sync::Arc::clone(&durability), ticket, + capture, submitted_at: std::time::Instant::now(), telemetry: self.telemetry.clone(), })) diff --git a/crates/cellule-runtime/src/publication/tests.rs b/crates/cellule-runtime/src/publication/tests.rs index 474a46b9..2b0f55d8 100644 --- a/crates/cellule-runtime/src/publication/tests.rs +++ b/crates/cellule-runtime/src/publication/tests.rs @@ -280,13 +280,14 @@ fn coverage_pending( Some(super::PendingDurability { durability: durability.clone(), ticket, + capture: None, submitted_at: std::time::Instant::now(), telemetry: Default::default(), }) } #[tokio::test] -async fn ordinary_pending_response_refuses_bundle_until_visibility_is_integrated() { +async fn ordinary_pending_response_refuses_bundle_without_original_capture_receipt() { use crate::node::log::DurabilitySource; let (durability, gate, _) = coverage_binding(1); let scratch = tempfile::tempdir().unwrap(); @@ -331,9 +332,7 @@ async fn ordinary_pending_response_refuses_bundle_until_visibility_is_integrated let pending = coverage_pending(&durability, ticket).unwrap(); assert!(matches!( pending.prove().await, - Err(Error::Node( - "shared bundle actor response path is not installed" - )) + Err(Error::Node("bundle response lacks original capture")) )); } @@ -802,6 +801,7 @@ async fn commits_report_when_no_enrolled_lane_can_carry_them() { timing: Default::default(), }; let submitter = CellDurabilitySubmitter { + binding: None, cell: CellId::from_bytes([71; 32]), incarnation: IncarnationId::from_bytes([72; 16]), epoch: 1, @@ -813,6 +813,7 @@ async fn commits_report_when_no_enrolled_lane_can_carry_them() { let lane: NodeDurabilitySlot = Arc::new(std::sync::RwLock::new(None)); let submitter = CellDurabilitySubmitter { + binding: None, node_durability: Some(lane), telemetry, ..submitter @@ -938,6 +939,7 @@ async fn commits_report_a_fenced_lane_instead_of_failing() { let submitter = CellDurabilitySubmitter { cell: CellId::from_bytes([84; 32]), + binding: None, incarnation: IncarnationId::from_bytes([85; 16]), epoch: 1, node_lease: None, diff --git a/docs/bundle-coverage-implementation.md b/docs/bundle-coverage-implementation.md index 9b4e45ca..8f2bf297 100644 --- a/docs/bundle-coverage-implementation.md +++ b/docs/bundle-coverage-implementation.md @@ -1,8 +1,9 @@ # Bundle coverage implementation The connected protocol APIs now implement shared selection, independently -awaitable root materialization and complete live-writer closure. They are **not -enabled in the ordinary actor response path**. The prior performance regression +awaitable root materialization and complete live-writer closure. An explicitly +installed original feed can now provide admitted receipts to the actor; +the example application's ordinary path still has no bundle producer. The prior performance regression and failed qualification remain the baseline. This slice establishes ordering and reconstruction evidence. The [fresh application-path benchmark](pr67-performance-reevaluation.md) measures `7fc0793`; it does not exercise bundle-based responses or establish @@ -53,9 +54,9 @@ Uploading an immutable object supplies no coverage proof. Live confirmation now reports `DurabilitySource::Bundle` separately from a materialized Cell root. Adjacent exact ranges merge into compact source intervals; root-only gaps keep their original source. A later exact root CAS -can still return an Object proof. Ordinary actors reject the Bundle source and -wait for their existing object fallback until command/read/retry visibility is -integrated. A boot without enrolled native log state supplies reconstruction +can still return an Object proof. Actors require the original admitted receipt, +assignment and Cell pin before accepting a Bundle response; raw local gate +confirmation alone cannot grant it. A boot without enrolled native log state supplies reconstruction proofs only. Root work superseded by confirmed coverage joins its old flusher before removing already-covered queue entries, including after lease loss. @@ -345,10 +346,37 @@ The experimental checkpoint API now takes its `BundleCoverageProof`, and Consumers must pass those original capabilities; a pin or sampled endpoint is not a substitute. The singleton checkpoint method delegates to the cohort path. -Remaining work before responses can use bundle proof: - -1. Integrate actor/executor proof and query/retry endpoints, release proved - capture retention, and schedule admitted materializers with joined shutdown. +## Selected capture cleanup in the actor + +The existing publisher now consumes a complete selected oldest capture prefix +before reading files for root preparation. The native worker checks original +scope, command endpoint, SQLite position, frame count and every captured segment +descriptor/body digest, and checks the live proof for each whole assignment. +It validates the complete prefix before deleting any local file. A matching +endpoint or a cold proof alone cannot release capture retention. + +The worker keeps outcomes and its selected proof pending. The actor drops its +duplicate capture indexes and returns their memory reservation while preserving +outcome admission. The same actor-owned publisher reconstructs from authenticated +origin locators, preserves the new command's due time, and uses ordinary root +lineage and Cell CAS. Storage retries retain the existing publication grace; +shutdown still joins materialization and complete original issued-range closure. +The installed feed gets a bounded 100-ms selection opportunity before the +ordinary root fallback. Root preparation already in progress retains its files. + +The real actor test blocks root preparation until selection, then pauses root +CAS and verifies that every selected capture file is absent. It issues a later +write and verifies that an unproven suffix refuses a query before its handler +runs; proving that exact later capture restores visibility. Both commands then +survive joined drain and cold restore. Separate cases retain prior Fleet ACKs +and accepted mutations after caller cancellation. These are ordering and +reconstruction tests, not TPS qualification. + +Remaining work before qualified production enablement: + +1. Install the canonical bounded node bundle producer and schedule fair admitted + materializer cohorts, coalescing proven debt rather than creating one root + per command. Preserve the integrated exact capture/visibility gate. 2. Build on the authenticated index: bound admitted maintenance inventory, increase checkpoint density with retained-byte accounting, and measure the complete materialization/checkpoint/collection cost. diff --git a/docs/pr67-bundle-receipt-measurement.md b/docs/pr67-bundle-receipt-measurement.md new file mode 100644 index 00000000..73f19e50 --- /dev/null +++ b/docs/pr67-bundle-receipt-measurement.md @@ -0,0 +1,126 @@ +# Current bundle receipt implementation: measured writes + +**Write parity is not achieved.** The frozen working tree completed 514.67 +Fleet writes/s and 225.77 Bucket writes/s in short Docker diagnostics. Fleet +was 2.3% below the retained baseline; Bucket was 8.78 times its baseline. +One overloaded repetition with workstation memory pressure cannot establish +a repeatable regression or an attributable improvement. + +## What was measured + +The candidate is working-tree code above `2dfbaa20223fa9473f9a25d4de16cc36f08f6941`, +including the admitted bundle receipt/actor integration and its assignment +sharing repair. It is **not an unmodified Git commit or the published PR head**. +The complete source manifest SHA-256 is +`fc9b1a91dd03ee11e726718ae3cf02d6fb8981555f450ed22d4eb5d627deba08`. +The measured Linux release SQL binary SHA-256 is +`864d68bec0eed92cf9c870d5b6f27ad2573ca72f7cd28d77dc17e2f13eba0af1`. +Baseline: `b1856728984781cee5398f8d7185fb88fde23993`. +Celld: pinned `f2bf648663a610eefde71f3547ad61e9b896b1f0` image. + +The example application does not install the new node bundle producer. +Bundle-proof and bundle-response histogram counts were zero. Existing shared +data packing operates, but ordinary responses and capture cleanup still depend +on follower proofs or per-Cell roots. These numbers measure the current real +application path; they do not measure an activated shared-selection ACK lane. + +All six cases used identical client/auditor binaries, fixture bytes, pinned +images, loaded runner, Docker host and point settings. Each had a fresh RustFS +volume. There were 1,000 uniformly active Cells, 96-byte values, 128 clients, +a 128-offer queue, SQL INSERT plus SELECT, and a two-hour request/result ledger. +This does not reproduce the laptop's bounded KV workload. + +## Environment and qualification limits + +The original 8-vCPU/16-GiB, 300-second profile completed one baseline window at +275.90 writes/s, then failed its warm audit. Its current-code run was interrupted +when host swap left less than 512 MiB free on macOS. That interrupted candidate +has no accepted TPS result. Its partial evidence and the original VM remain +retained. The qualification profile and expected evidence remain unchanged. + +The completed measurements below are explicitly a separate **diagnostic**: +one 60-second window after 30 seconds of warmup, on an ARM64 Linux VM with +8 vCPUs and nominal 8 GiB **total** RAM. Owner, two Fleet followers, RustFS and +client share that VM. Node container ceilings remain 8 CPUs/16 GiB, with +4-GiB tmpfs state; these ceilings exceed the available shared memory. +RustFS has a 2-CPU/8-GiB ceiling and the client 4 CPUs/4 GiB. +Cellule retains its matched 64-MiB retained-memory and 1-GiB managed-disk budgets. +The physical workstation also runs another VM and experienced host memory/swap +pressure. These observations establish neither dedicated-node capacity nor +physical-device durability. No separate read or mixed workload was measured. + +## Independently reconciled windows + +TPS counts successful logical writes completed inside the timed window. +Latency is the exact nearest-rank percentile of measured successful attempts, +including trailing successful completions. Scheduled latency starts at the +offered arrival; request latency starts when sent. Errors and drops are separate. + +| Mode / system | Successful writes/s | Successful scheduled p99 ms | Successful request p99 ms | Errors | Queue drops | +| --- | ---: | ---: | ---: | ---: | ---: | +| Fleet / baseline | 526.90 | 219.92 | 145.74 | 753,964 | 114,422 | +| Fleet / current code | 514.67 | 144.77 | 107.99 | 716,143 | 152,977 | +| Fleet / celld | 3,949.52 | 146.95 | 78.92 | 0 | 662,773 | +| Bucket / baseline | 25.70 | 36,560.60 | 14,049.79 | 0 | 118,202 | +| Bucket / current code | 225.77 | 4,856.26 | 4,159.45 | 0 | 106,198 | +| Bucket / celld | 861.78 | 1,553.14 | 758.28 | 0 | 68,038 | + +Fleet offered 15,000 writes/s; Bucket offered 2,000/s. The original offers, +attempts, errors, successes, trailing successes and drops reconcile exactly. +Current Fleet had 30,880 successes inside its window and no late successes. +Current Bucket had 13,546 inside and 256 after it. None passes the unchanged +delivery/latency targets or establishes sustainable maximum throughput. + +Both Cellule arms passed all-ACK warm reads/retries and bucket-only cold +recovery in both modes. Current Fleet checked 53,045 ACKs and drained in 6.17 s; +current Bucket checked 21,805 and drained in 7.70 s. These cohorts include setup +and warmup. Celld Bucket checked 84,460 ACKs and passed both audits. Celld Fleet +failed the warm audit with 330,004 errors among 394,221 ACKs; cold recovery was +not reached. Its throughput is an observed window rate, not qualified capacity. +The audit failures establish unavailable checks, not proven data loss. + +## Attempted second repetition + +A second diagnostic attempted the same immutable binaries and workload with +the baseline before the candidate. Its Fleet baseline completed 29,539 writes +inside 60 seconds: **492.32 writes/s**, successful scheduled p99 **181.72 ms**, +696,148 HTTP 503 errors and 174,313 queue drops. Counts reconcile to 900,000 +offers. All 50,867 ACKs, including setup and warmup, passed both warm and cold +read/retry audits; drain took 7.53 seconds. + +The runner then refused to start the candidate because host disk headroom had +fallen below its unchanged 4-GiB pre-case boundary as workstation swap grew. +The benchmark VM was stopped. Only one of six planned cases was attempted; +there is **no second candidate or celld measurement**, no second paired result, +and no basis for treating the first pair's difference as a repeatable gain. +The independent journal replay verified this baseline window and the frozen +source; qualification remains false. + +Its external evidence label is `bundle-receipt-r2-vm8c-20261008`; the evidence +index SHA-256 is +`7d367ee810099d30a13ae6cb13d44328fb4bc50c370d56a03a972c7bd1a5b221`. + +In the completed first comparison, current Fleet selected 13,201 roots, +averaging 2.38 materialized commits/root; +Bucket selected 13,643, averaging 1.01. Window-only successful provider PUTs per +completed command were 1.72 and 3.60 respectively. These exclude trailing work +and SDK-internal retries. Per-Cell publication remains sparse; the design's +dense checkpoint and admitted materializer goals have not been demonstrated. + +## Verification and evidence + +The frozen source passed all 11 contributor verification routes: 1,949 workspace +tests passed, with 38 documented ignored tests. An independent journal replay +verified all six window counts, successful percentiles, ACK stream hashes, +source manifests, adapted build cache key and measured binary hashes. Provider +capacity and lifecycle samples remained valid. This verifies measurement +integrity, not qualification. Three paired five-minute repetitions, the +16-GiB serving-node profile, bounded stable debt, read guardrails and remaining +[design exit gates](../crates/cellule-runtime/docs/write-performance-design.md) +remain outstanding. + +Raw evidence stays outside Git under +`/Volumes/Workspace/crabbuild-target/cellule-write-perf-8ad1`, label +`bundle-receipt-r2-vm8b-20261008`. The evidence index SHA-256 is +`c7f3e02daa28f100ac1b1df233357842dfe5a4f32da4d27c419d4564288200b8`. +Failed preflight attempts and interrupted runs are retained separately. From 7b986b05b440d1501bac906ebcb42b8efd25fffc Mon Sep 17 00:00:00 2001 From: forhappy Date: Thu, 8 Oct 2026 11:45:05 -0700 Subject: [PATCH 035/102] Record measured selected-capture TPS and failed parity gates --- docs/pr67-bundle-receipt-measurement.md | 6 +- ...67-selected-capture-release-measurement.md | 132 ++++++++++++++++++ docs/write-performance-delivery.md | 12 +- 3 files changed, 148 insertions(+), 2 deletions(-) create mode 100644 docs/pr67-selected-capture-release-measurement.md diff --git a/docs/pr67-bundle-receipt-measurement.md b/docs/pr67-bundle-receipt-measurement.md index 73f19e50..8b64a42e 100644 --- a/docs/pr67-bundle-receipt-measurement.md +++ b/docs/pr67-bundle-receipt-measurement.md @@ -1,4 +1,8 @@ -# Current bundle receipt implementation: measured writes +# Bundle receipt implementation: historical measured writes + +The [selected-capture release measurement](pr67-selected-capture-release-measurement.md) +records later committed code at `3e43601`. The working-tree results below measure +the earlier receipt snapshot and must not be attributed to that commit. **Write parity is not achieved.** The frozen working tree completed 514.67 Fleet writes/s and 225.77 Bucket writes/s in short Docker diagnostics. Fleet diff --git a/docs/pr67-selected-capture-release-measurement.md b/docs/pr67-selected-capture-release-measurement.md new file mode 100644 index 00000000..245ac364 --- /dev/null +++ b/docs/pr67-selected-capture-release-measurement.md @@ -0,0 +1,132 @@ +# Selected bundle capture release: measured writes + +**Write parity is not achieved.** Current code completed +**461.53 Fleet writes/s** and **168.92 Bucket writes/s**. +Fleet was 5.9% higher than baseline; Bucket was 18.2% lower. +This report measures committed runtime code +`3e436013a7e7500694bfdfe5600fe4a69adb1988`, including exact selected-capture +release through the actor-owned publisher. One short overloaded repetition +cannot establish a repeatable improvement or regression. The embedding SQL +application still does not install a node bundle producer, so the new shared +selection response/capture-release lane was not exercised by these TPS cases. + +## Matched diagnostic + +The six cases ran serially on the same ARM64 Docker VM with **8 vCPUs and +8 GiB total RAM**, shared by the serving nodes, RustFS and client. Fleet used +one owner and two followers. Each case used a fresh object-store volume. +Node container ceilings were 8 CPUs/16 GiB with 4-GiB tmpfs state; RustFS had +a 2-CPU/8-GiB ceiling and the client 4 CPUs/4 GiB. These ceilings exceed the +VM's available memory. Cellule retained its 64-MiB retained-memory and 1-GiB +managed-disk budgets. Another workstation VM remained running. + +All arms used identical load/audit binaries, fixture bytes, pinned images and +runner code: 1,000 uniform Cells, 96-byte values, 128 clients and 128 queue +slots, SQL INSERT plus SELECT and a two-hour request/result ledger. Each point +had 30 seconds of warmup and a 60-second measured window. Fleet offered 15,000 +writes/s; Bucket offered 2,000/s. This differs from the laptop's bounded KV +workload and the dedicated 8-vCPU/16-GiB serving-node qualification. There +were no separate read-only or mixed measurements. + +Baseline: `b1856728984781cee5398f8d7185fb88fde23993`. +Celld: `f2bf648663a610eefde71f3547ad61e9b896b1f0`. + +## Independently reconciled windows + +TPS counts successful logical writes completed inside the window. Successful +latency is the exact nearest-rank percentile replayed from request journals, +including trailing successful completions. Scheduled latency starts at the +offered arrival; request latency starts at send. Errors and drops are excluded +from successful percentiles and remain separate delivery failures. + +| Mode / system | Successful writes/s | Successful scheduled p99 ms | Successful request p99 ms | Errors | Queue drops | +| --- | ---: | ---: | ---: | ---: | ---: | +| Fleet / baseline | 435.62 | 302.84 | 149.11 | 680,024 | 193,839 | +| Fleet / current code | 461.53 | 266.49 | 192.42 | 685,202 | 187,106 | +| Fleet / celld | 3,525.58 | 204.00 | 113.85 | 0 | 688,209 | +| Bucket / baseline | 206.50 | 4,979.39 | 4,381.86 | 0 | 107,354 | +| Bucket / current code | 168.92 | 10,784.35 | 8,705.00 | 0 | 109,609 | +| Bucket / celld | 486.93 | 1,754.17 | 961.24 | 0 | 90,528 | + +All generated offers reconcile to successful, errored, dropped or unissued +offers. None passes the unchanged Fleet 15K TPS/p99 50-ms or Bucket 2K TPS/ +p99 200-ms delivery gates with zero errors and drops. These are observed +overloaded window rates, not sustainable maximum throughput. + +| Mode | Current-code successful writes inside / after window | Current / baseline observed TPS | +| --- | ---: | ---: | +| Fleet | 27,692 / 0 | 1.059× | +| Bucket | 10,135 / 256 | 0.818× | + +The ratios describe this single pair only. Inactive bundle counters prevent +attribution to the new capture-release lane. Workstation conditions and run +order remain confounders; older snapshots' measurements remain separate. + +## Recovery and publication observations + +| Mode / system | ACK cohort | Warm / cold checks | Warm / cold | Owner drain seconds | +| --- | ---: | --- | --- | ---: | +| Fleet / baseline | 45,266 | 45,266 / 45,266 | pass / pass | 11.06 | +| Fleet / current code | 47,369 | 47,369 / 47,369 | pass / pass | 5.85 | +| Fleet / celld | 356,306 | 356,306 / 356,306 | pass / pass | 25.39 | +| Bucket / baseline | 19,657 | 19,657 / 19,657 | pass / pass | 10.39 | +| Bucket / current code | 20,391 | 20,391 / 20,391 | pass / pass | 28.04 | +| Bucket / celld | 50,585 | 50,585 / 50,585 | pass / pass | 3.24 | + +ACK cohorts include setup, warmup, steady and trailing successful commands. +Cold recovery reconstructs from the bucket with the old local state and +follower logs unavailable; retries check original outcomes. These successful +audits do not qualify power-loss durability on physical storage. + +| Current-code mode | Bundle proof / response count | Selected Cell roots | Commits / root | Successful provider PUTs / completed command | +| --- | ---: | ---: | ---: | ---: | +| Fleet | 0 / 0 | 12,174 | 2.31 | 1.75 | +| Bucket | 0 / 0 | 10,241 | 1.01 | 3.64 | + +Provider totals are window-only storage API observations, excluding trailing +publication and SDK-internal retries. Per-Cell roots remain sparse. Fleet +retained memory was near its 64-MiB ceiling at the initial sample and all eight +dirty publication slots were occupied at both boundaries. This is evidence +of backlog/admission pressure, not an isolated causal profile. Shared capture +upload batching does not remove the remaining per-Cell authority work. + +## Delivered code and remaining gates + +The actor can now release an exact selected capture prefix before its Cell-root +CAS and rebuild that root from the authenticated selected origin through its +existing serialized publisher. Original assignment fingerprints cover every +ordered segment descriptor and body digest. The executor retains outcomes and +root obligations until joined materialization; an unproved newer suffix remains +hidden. A regression test pauses root selection, verifies capture files and +metadata were released, adds a newer command, then checks drain, retry, cold +restore and complete resource release. No wire or persisted format changed. + +A canonical bounded node bundle producer and fair admitted materializer +scheduling still need framework implementation and host integration before +improving the measured application path. Dense checkpoints, stable bounded debt, complete +failed-owner issued-suffix orchestration and safe cross-Cell collection remain +open. Three paired five-minute runs, the 16-GiB serving-node profile, 2,000-Cell +10K-write/50K-read qualification and read guardrails remain outstanding. + +## Verification and external evidence + +The measured source passed all 11 contributor verification routes in an +isolated snapshot: **1,951 workspace tests passed, zero failed, 38 documented +ignored**. An independent journal replay verified all six window counts, +successful percentiles, ACK hashes/cohorts, source and adapted build manifests, +matched binaries and provider lifecycle/capacity evidence. Measurement integrity +passed; performance qualification remains false. + +The complete source-manifest SHA-256 is +`e4ad9aea226ede251ae88ffa5cfcde086aacda80823d151c26df19736ec41548`. +The measured Linux SQL release binary SHA-256 is +`96546154c9d2b327d6ad9d0a0cd1ea02e3a73d6613e07a4ff529c5158a20edda`. + +Raw journals, manifests, release binaries and logs remain outside Git under +`/Volumes/Workspace/crabbuild-target/cellule-write-perf-8ad1`, evidence label +`selected-capture-release-3e43601-vm8-20261008`. The evidence-index SHA-256 is +`c6435776c737465303a2e7a31547ed6a9136048e7d399da2f7eea214af219ecc`. +The earlier [receipt measurement](pr67-bundle-receipt-measurement.md) and its +interrupted attempts remain historical evidence. See the +[runtime design](../crates/cellule-runtime/docs/write-performance-design.md) +for the unchanged exit gates. diff --git a/docs/write-performance-delivery.md b/docs/write-performance-delivery.md index cff87a12..2ecd4251 100644 --- a/docs/write-performance-delivery.md +++ b/docs/write-performance-delivery.md @@ -6,7 +6,17 @@ comparison retains exact retry and cold-state audits. **Celld write parity has not been established.** The [proposal](write-performance-proposal.md) remains the acceptance contract; completing tests or a load run does not pass its gates. -The [PR 67 reevaluation](pr67-performance-reevaluation.md) measures the latest +The latest [selected-capture release measurement](pr67-selected-capture-release-measurement.md) +records committed code at `3e43601`: 461.53 Fleet writes/s and 168.92 Bucket +writes/s in one matched 60-second Docker diagnostic. Fleet was 5.9% higher than +baseline; Bucket was 18.2% lower, with worse successful-write latency. All ACK +audits passed, but every system failed delivery targets. The actor can release +exact selected captures through its original publisher; the application still +does not install a node bundle producer, and bundle response counters are zero. +These results establish neither an attributable improvement nor parity. The +older comparisons below remain historical evidence. + +The [PR 67 reevaluation](pr67-performance-reevaluation.md) measures the earlier protocol implementation at `7fc0793` in nine fresh matched Docker cases. Fleet 100/s p99 is 19.9 ms versus main's 34.7 ms and celld's 16.0 ms. Target-load delivery still fails: candidate Fleet completion is below main, bucket is From 2dc19172ce71d1f7b051705d70f005a70a47e554 Mon Sep 17 00:00:00 2001 From: forhappy Date: Thu, 8 Oct 2026 12:51:59 -0700 Subject: [PATCH 036/102] Retain and join bounded node bundle publication --- .../cellule-axum/examples/fleet/authority.rs | 140 +++++++ crates/cellule-axum/examples/fleet/mod.rs | 4 +- .../cellule-runtime/src/cell/actor/bundle.rs | 7 +- .../src/cell/actor/requests.rs | 6 + .../src/node/bundle/selection.rs | 55 +++ .../src/node/bundle/tests/actor.rs | 66 ++- .../src/node/bundle/tests/managed.rs | 376 ++++++++++++++++++ .../src/node/bundle/tests/mod.rs | 1 + .../src/node/bundle/tests/receipts.rs | 24 ++ .../src/node/durability/mod.rs | 90 ++++- .../src/node/durability/publication/mod.rs | 335 ++++++++++++++++ crates/cellule-runtime/src/node/log/mod.rs | 4 + .../src/node/log_shipper/publication/mod.rs | 4 + crates/cellule-runtime/src/publication/mod.rs | 37 ++ 14 files changed, 1140 insertions(+), 9 deletions(-) create mode 100644 crates/cellule-runtime/src/node/bundle/tests/managed.rs create mode 100644 crates/cellule-runtime/src/node/durability/publication/mod.rs diff --git a/crates/cellule-axum/examples/fleet/authority.rs b/crates/cellule-axum/examples/fleet/authority.rs index b8467239..a4f69b41 100644 --- a/crates/cellule-axum/examples/fleet/authority.rs +++ b/crates/cellule-axum/examples/fleet/authority.rs @@ -1,6 +1,9 @@ //! One serialized directory view for heartbeats and durability CAS callbacks. use super::*; use cellule_runtime::node::durability::NodeLogAuthority; +use cellule_runtime::node::durability::{ + BundleCheckpoint, NodeBundleAuthority, NodeBundlePublicationAuthority, +}; use cellule_runtime::node::log::{NodeLogRetirementObservation, NodeLogRotationBarrier}; use cellule_runtime::node::{ NodeAdvertisement, NodeCapacity, NodeFailureDomain, VersionedNodeAdvertisement, @@ -151,6 +154,10 @@ impl Authority { .directory .recruit_log(&state.observed, 1, 16 << 20, 3, clock()?) .await?; + state.observed = self + .directory + .initialize_bundle_lane(&state.observed, 1, clock()?) + .await?; let members = state .observed .advertisement() @@ -199,6 +206,139 @@ impl Authority { } } +impl NodeBundleAuthority for Authority { + fn bind<'a>( + &'a self, + authority: &'a cellule_runtime::control::authority::CellAuthority, + observed: &'a cellule_runtime::control::authority::VersionedControl, + ) -> BoxFuture<'a, Result> { + Box::pin(async move { + let mut state = self.state.lock().await; + let current = self.current(&state, 1).await?; + let (next, pinned) = self + .directory + .bind_bundle_cell(¤t, authority, observed, clock()?) + .await?; + state.observed = next; + self.lease.check()?; + Ok(pinned) + }) + } + + fn close<'a>( + &'a self, + authority: &'a cellule_runtime::control::authority::CellAuthority, + observed: &'a cellule_runtime::control::authority::VersionedControl, + issued: cellule_runtime::node::log::CellIssuedRange, + ) -> BoxFuture<'a, Result<()>> { + Box::pin(async move { + // Runtime joins the complete issued producer prefix before entering + // this mutex, including captures whose Fleet ACK preceded selection. + let mut state = self.state.lock().await; + let mut current = self.current(&state, issued.log_epoch()).await?; + let proof = self + .directory + .load_bundle_coverage(authority, observed, cellule_ltx::Limits::default()) + .await?; + current = self + .directory + .checkpoint_bundle_cell( + ¤t, + authority, + &proof, + cellule_ltx::Limits::default(), + clock()?, + ) + .await?; + state.observed = current; + state.observed = self + .directory + .begin_bundle_close(&state.observed, proof.binding(), issued, clock()?) + .await?; + state.observed = self + .directory + .finish_bundle_close(&state.observed, proof.binding(), issued, clock()?) + .await?; + self.lease.check() + }) + } +} + +impl NodeBundlePublicationAuthority for Authority { + fn select<'a>( + &'a self, + captures: &'a [cellule_runtime::node::log_shipper::AssignedCapture], + lease: &'a NodeLeaseGuard, + ) -> BoxFuture<'a, Result>> { + Box::pin(async move { + let mut state = self.state.lock().await; + let current = self.current(&state, 1).await?; + let frames = captures + .iter() + .flat_map(|c| c.frames().iter().cloned()) + .collect::>(); + let assignments = captures.iter().map(|c| c.assignment()).collect::>(); + let prepared = self + .directory + .prepare_node_bundle(¤t, &frames, &assignments, clock()?) + .await?; + let (next, proofs) = self + .directory + .select_node_bundle( + ¤t, + &prepared, + lease, + cellule_ltx::Limits::default(), + clock()?, + ) + .await?; + state.observed = next; + self.lease.check()?; + Ok(proofs) + }) + } + + fn checkpoint<'a>(&'a self, checkpoints: &'a [BundleCheckpoint]) -> BoxFuture<'a, Result<()>> { + Box::pin(async move { + let mut state = self.state.lock().await; + let current = self.current(&state, 1).await?; + let mut ready = Vec::new(); + for checkpoint in checkpoints { + let root = checkpoint.root(); + let observed = checkpoint + .authority() + .load(cellule_runtime::CellId::from_bytes(root.cell)) + .await? + .ok_or(Error::Fenced)?; + let actual = observed.value().ltx_root().ok_or(Error::Fenced)?; + if observed.value().bundle_binding == Some(checkpoint.proof().binding()) + && actual.cell == root.cell + && actual.incarnation == root.incarnation + && actual.commit_sequence > root.commit_sequence + { + // The later root's original notification is retained by its + // joined publisher. This observation releases no locators. + continue; + } + ready.push((checkpoint.authority(), checkpoint.proof())); + } + state.observed = if ready.is_empty() { + current + } else { + self.directory + .checkpoint_bundle_cells( + ¤t, + &ready, + cellule_ltx::Limits::default(), + clock()?, + ) + .await? + }; + self.lease.check() + }) + } +} + fn capacity(follower: Option<&cellule_runtime::follower::FollowerStore>) -> NodeCapacity { // Memory/disk hints are fixture admission ceilings, not measured physical // headroom. Follower retention and free budget come from the canonical store. diff --git a/crates/cellule-axum/examples/fleet/mod.rs b/crates/cellule-axum/examples/fleet/mod.rs index b0df32f1..12431b4f 100644 --- a/crates/cellule-axum/examples/fleet/mod.rs +++ b/crates/cellule-axum/examples/fleet/mod.rs @@ -136,7 +136,9 @@ impl Config { cellule_ltx::Limits::default(), runtime.telemetry_handle(), )?; - runtime.install_node_durability(application, config.build()?)?; + let durability = config.build()?; + runtime.install_node_durability(application, durability.clone())?; + durability.start_bundle_publication(enrollment.authority.clone())?; Ok(()) } .await; diff --git a/crates/cellule-runtime/src/cell/actor/bundle.rs b/crates/cellule-runtime/src/cell/actor/bundle.rs index a26c29bb..a62163c3 100644 --- a/crates/cellule-runtime/src/cell/actor/bundle.rs +++ b/crates/cellule-runtime/src/cell/actor/bundle.rs @@ -26,7 +26,7 @@ pub(super) async fn selected_prefix( let mut captures = Vec::with_capacity(durabilities.len()); for pending in durabilities.iter().flatten() { let capture = pending - .selected_capture() + .selected_capture_prefix() .await? .ok_or(Error::Control("bundle selection lost original capture"))?; captures.push(capture); @@ -41,9 +41,8 @@ pub(super) async fn selected_prefix( return Ok(None); }; let proof = &newest.selected().proof; - // A proof can cover later commands outside this task's retained cohort. - // Keep the root fallback in that case; endpoint equality alone cannot - // authorize pruning captures that this publisher does not own. + // An original whole-capture capability narrows a later cohort's proof to + // this publisher's exact endpoint; an arbitrary watermark cannot do so. if proof.commit_sequence() != commit_sequence || proof.position() != position { return Ok(None); } diff --git a/crates/cellule-runtime/src/cell/actor/requests.rs b/crates/cellule-runtime/src/cell/actor/requests.rs index a31b3247..24c56a6a 100644 --- a/crates/cellule-runtime/src/cell/actor/requests.rs +++ b/crates/cellule-runtime/src/cell/actor/requests.rs @@ -590,6 +590,9 @@ pub(super) fn start_admitted_publication( }; pool.bind_bundle_materialized(cell, root).await?; PendingDurability::prove_objects(&durabilities).await?; + if let Some(pending) = durabilities.last().and_then(Option::as_ref) { + pending.checkpoint_materialized(publisher.authority(), root).await?; + } let published = pool.confirm_published_range(cell, root).await?; authority = authority_started.elapsed(); if published != expected { @@ -671,6 +674,9 @@ pub(super) fn start_admitted_publication( }; authority = authority_started.elapsed(); let logged = PendingDurability::prove_objects(&durabilities).await?; + if let Some(pending) = durabilities.last().and_then(Option::as_ref) { + pending.checkpoint_materialized(publisher.authority(), root).await?; + } if !logged { publisher.record_object_proof(newest_submitted_at.elapsed()); } diff --git a/crates/cellule-runtime/src/node/bundle/selection.rs b/crates/cellule-runtime/src/node/bundle/selection.rs index 728d70f9..c29420aa 100644 --- a/crates/cellule-runtime/src/node/bundle/selection.rs +++ b/crates/cellule-runtime/src/node/bundle/selection.rs @@ -3,6 +3,7 @@ use super::*; use crate::node::lease::NodeLeaseGuard; use crate::node::{NodeDirectory, VersionedNodeAdvertisement}; +#[derive(Clone)] pub(super) struct LiveBundleCoverage { node: crate::identity::NodeId, lease: NodeLeaseGuard, @@ -37,6 +38,60 @@ impl LiveBundleCoverage { } } +impl BundleCoverageProof { + /// Narrows an original selected cohort to one complete original capture, + /// retaining the authenticated historical prefix and the original lease. + /// An endpoint alone cannot create this capability. + pub(crate) fn original_capture_prefix( + &self, + assignment: &crate::node::log::AssignedCommitRange, + ) -> Result { + self.check_live_assignment(assignment)?; + let live = self.live.as_ref().ok_or(Error::Fenced)?; + let index = live + .assignments + .iter() + .position(|a| a == assignment) + .ok_or(Error::Node("bundle lacks original assigned capture"))?; + let count = |a: &crate::node::log::AssignedCommitRange| { + usize::try_from(a.ticket().last_sequence() - a.ticket().first_sequence() + 1) + .map_err(|_| Error::Capacity("bundle capture prefix")) + }; + let frames = live.assignments.iter().try_fold(0_usize, |n, a| { + n.checked_add(count(a)?) + .ok_or(Error::Capacity("bundle capture prefix")) + })?; + let retained = live.assignments[..=index] + .iter() + .try_fold(0_usize, |n, a| { + n.checked_add(count(a)?) + .ok_or(Error::Capacity("bundle capture prefix")) + })?; + let historical = self + .binding + .locators + .len() + .checked_sub(frames) + .ok_or(Error::Node("bundle capture prefix omits assigned frames"))?; + let mut binding = self.binding.clone(); + binding.locators.truncate(historical + retained); + let (first, commit, position) = assignment.endpoint(); + binding.first_commit = first; + binding.selected_commit = commit; + binding.selected_position = position; + binding.selected_sequence = assignment.ticket().last_sequence(); + let mut live = live.clone(); + live.assignments.truncate(index + 1); + Ok(Self { + pin: self.pin, + binding, + head: self.head, + session: self.session, + live: Some(live), + }) + } +} + impl NodeDirectory { /// Verifies a contiguous complete native range and uploads one proposal for /// every participating Cell. Neither upload nor this value grants an ACK. diff --git a/crates/cellule-runtime/src/node/bundle/tests/actor.rs b/crates/cellule-runtime/src/node/bundle/tests/actor.rs index 3d783b8d..789bbf10 100644 --- a/crates/cellule-runtime/src/node/bundle/tests/actor.rs +++ b/crates/cellule-runtime/src/node/bundle/tests/actor.rs @@ -6,7 +6,10 @@ use crate::cell::executor::{HandlerOutcome, MutationIdentity, Resolution}; use crate::cell::worker::SqlWorkerPool; use crate::fleet::telemetry::{CellTelemetry, CommandResponseSource}; use crate::identity::{CellTarget, NamespaceId, RequestId, TenantId}; -use crate::node::durability::{NodeBundleAuthority, NodeDurability, NodeLogAuthority}; +use crate::node::durability::{ + BundleCheckpoint, NodeBundleAuthority, NodeBundlePublicationAuthority, NodeDurability, + NodeLogAuthority, +}; use crate::node::log_shipper::NodeLogShipper; use crate::node::log_transport::{ AppendRequest, LocalFollowerTransport, NodeLogTransport, RetireRequest, SealRequest, @@ -102,10 +105,67 @@ impl NodeLogAuthority for Authority { } } +impl NodeBundlePublicationAuthority for Authority { + fn select<'a>( + &'a self, + captures: &'a [crate::node::log_shipper::AssignedCapture], + lease: &'a NodeLeaseGuard, + ) -> BoxFuture<'a, Result>> { + Box::pin(async move { + let mut node = self.observed.lock().await; + let frames = captures + .iter() + .flat_map(|c| c.frames().iter().cloned()) + .collect::>(); + let assignments = captures.iter().map(|c| c.assignment()).collect::>(); + let prepared = self + .directory + .prepare_node_bundle(&node, &frames, &assignments, NOW) + .await?; + let (next, proofs) = self + .directory + .select_node_bundle(&node, &prepared, lease, Limits::default(), NOW) + .await?; + *node = next; + Ok(proofs) + }) + } + + fn checkpoint<'a>(&'a self, checkpoints: &'a [BundleCheckpoint]) -> BoxFuture<'a, Result<()>> { + Box::pin(async move { + let mut node = self.observed.lock().await; + let mut ready = Vec::new(); + for checkpoint in checkpoints { + let current = checkpoint + .authority() + .load(CellId::from_bytes(checkpoint.root().cell)) + .await? + .ok_or(Error::Fenced)?; + let root = current.value().ltx_root().ok_or(Error::Fenced)?; + if current.value().bundle_binding == Some(checkpoint.proof().binding()) + && root.cell == checkpoint.root().cell + && root.incarnation == checkpoint.root().incarnation + && root.commit_sequence > checkpoint.root().commit_sequence + { + continue; + } + ready.push((checkpoint.authority(), checkpoint.proof())); + } + if !ready.is_empty() { + *node = self + .directory + .checkpoint_bundle_cells(&node, &ready, Limits::default(), NOW) + .await?; + } + Ok(()) + }) + } +} + pub(super) struct PausedFollowers { peers: [LocalFollowerTransport; 2], - released: AtomicBool, - changed: tokio::sync::Notify, + pub(super) released: AtomicBool, + pub(super) changed: tokio::sync::Notify, } pub(super) fn transport(f: &Fixture, released: bool) -> Arc { diff --git a/crates/cellule-runtime/src/node/bundle/tests/managed.rs b/crates/cellule-runtime/src/node/bundle/tests/managed.rs new file mode 100644 index 00000000..9d073f71 --- /dev/null +++ b/crates/cellule-runtime/src/node/bundle/tests/managed.rs @@ -0,0 +1,376 @@ +//! Ordinary actors driven by the runtime producer, without a manual selector. +use super::actor::{Authority, transport}; +use super::*; +use crate::cell::actor::CellRuntime; +use crate::cell::catalog::{CatalogEntry, CatalogRole, CellCatalog}; +use crate::cell::executor::{HandlerOutcome, MutationIdentity}; +use crate::cell::worker::SqlWorkerPool; +use crate::fleet::telemetry::{CellTelemetry, CommandResponseSource}; +use crate::identity::{CellTarget, NamespaceId, RequestId, TenantId}; +use crate::node::durability::NodeDurability; +use crate::node::durability::{ + BundleCheckpoint, NodeBundleAuthority, NodeBundlePublicationAuthority, +}; +use crate::node::log_shipper::NodeLogShipper; +use crate::node::log_shipper::{AssignedCapture, NodeLogSubmission}; +use futures_util::future::BoxFuture; +use std::sync::{Mutex, atomic::Ordering}; + +#[derive(Default)] +struct Responses(Mutex>); +impl CellTelemetry for Responses { + fn command_response( + &self, + source: CommandResponseSource, + _: std::time::Duration, + _: std::time::Duration, + ) { + self.0.lock().unwrap().push(source); + } +} + +#[tokio::test] +async fn rejected_producer_working_credit_does_not_install_an_irreversible_feed() { + let mut f = Fixture::new().await; + super::coverage::enroll(&mut f).await; + let authority = Arc::new(Authority { + directory: f.directory.clone(), + observed: tokio::sync::Mutex::new(f.node.clone()), + }); + let peers = transport(&f, true); + let shipper = NodeLogShipper::new(f.gate.clone(), peers.clone(), Limits::default()).unwrap(); + let durability = Arc::new(NodeDurability::new( + f.gate.clone(), + shipper, + authority.clone(), + peers, + f.lease.clone(), + )); + let pool = SqlWorkerPool::new(1, 1).unwrap(); + pool.configure_retained_capacity(8 << 20).unwrap(); + durability + .attach_selection_resources(pool.resource_ledger()) + .unwrap(); + assert!(matches!( + durability.start_bundle_publication(authority), + Err(Error::Capacity(_)) + )); + let mut feed = durability.take_publication_feed().unwrap(); + durability.shutdown().await.unwrap(); + assert!(feed.recv().await.is_none()); + assert_eq!( + pool.resource_ledger() + .snapshot() + .unwrap() + .used + .retained_bytes(), + 0 + ); + pool.shutdown().await.unwrap(); +} + +struct FailedSelection(Arc); +impl NodeBundleAuthority for FailedSelection { + fn bind<'a>( + &'a self, + authority: &'a CellAuthority, + observed: &'a VersionedControl, + ) -> BoxFuture<'a, Result> { + NodeBundleAuthority::bind(self.0.as_ref(), authority, observed) + } + fn close<'a>( + &'a self, + authority: &'a CellAuthority, + observed: &'a VersionedControl, + issued: crate::node::log::CellIssuedRange, + ) -> BoxFuture<'a, Result<()>> { + NodeBundleAuthority::close(self.0.as_ref(), authority, observed, issued) + } +} +impl NodeBundlePublicationAuthority for FailedSelection { + fn select<'a>( + &'a self, + _: &'a [AssignedCapture], + _: &'a NodeLeaseGuard, + ) -> BoxFuture<'a, Result>> { + Box::pin(async { Err(Error::Node("injected bundle selection failure")) }) + } + fn checkpoint<'a>(&'a self, checkpoints: &'a [BundleCheckpoint]) -> BoxFuture<'a, Result<()>> { + self.0.checkpoint(checkpoints) + } +} + +#[tokio::test] +async fn producer_failure_fences_new_work_and_join_preserves_its_cause() { + let mut f = Fixture::new().await; + super::coverage::enroll(&mut f).await; + let mut cell = f.cell(4).await; + let authority = Arc::new(Authority { + directory: f.directory.clone(), + observed: tokio::sync::Mutex::new(f.node.clone()), + }); + let peers = transport(&f, true); + let shipper = NodeLogShipper::new(f.gate.clone(), peers.clone(), Limits::default()).unwrap(); + let durability = Arc::new(NodeDurability::new( + f.gate.clone(), + shipper, + authority.clone(), + peers, + f.lease.clone(), + )); + let pool = SqlWorkerPool::new(1, 1).unwrap(); + pool.configure_retained_capacity(32 << 20).unwrap(); + durability + .attach_selection_resources(pool.resource_ledger()) + .unwrap(); + durability + .start_bundle_publication(Arc::new(FailedSelection(authority))) + .unwrap(); + cell.db + .transaction(|tx| tx.execute("INSERT INTO outcomes VALUES('failed','retained')", [])) + .unwrap(); + let cuts = cell.db.capture().unwrap(); + durability + .submit_assigned( + NodeLogSubmission::new( + ApplicationId::from_bytes([9; 16]), + cell.control.value().cell, + cell.control.value().incarnation, + cell.control.value().epoch, + 2, + &cuts, + ) + .unwrap(), + ) + .await + .unwrap(); + tokio::time::timeout(std::time::Duration::from_secs(5), f.lease.wait_fenced()) + .await + .unwrap(); + let error = durability.shutdown().await.unwrap_err(); + assert!( + matches!(error, Error::Shared(source) if matches!(source.as_ref(), Error::Node("injected bundle selection failure"))) + ); + assert_eq!(durability.progress().unwrap().tiered_through, 0); + assert_eq!(durability.progress().unwrap().issued_through, 1); + assert_eq!( + pool.resource_ledger() + .snapshot() + .unwrap() + .used + .retained_bytes(), + 0 + ); + pool.shutdown().await.unwrap(); +} + +#[tokio::test(flavor = "multi_thread")] +async fn managed_producer_selects_actor_prefixes_and_joins_checkpoints_complete_close_and_cold_results() + { + let mut f = Fixture::new().await; + super::coverage::enroll(&mut f).await; + let authority = Arc::new(Authority { + directory: f.directory.clone(), + observed: tokio::sync::Mutex::new(f.node.clone()), + }); + let peers = transport(&f, false); + let shipper = NodeLogShipper::new(f.gate.clone(), peers.clone(), Limits::default()).unwrap(); + let durability = Arc::new(NodeDurability::new( + f.gate.clone(), + shipper, + authority.clone(), + peers.clone(), + f.lease.clone(), + )); + let pool = SqlWorkerPool::new(2, 4).unwrap(); + let dirty = Arc::new(tokio::sync::Semaphore::new(1)); + let runtime = CellRuntime::new_with_replica_host_requiring_node_lease( + pool.clone(), + 32 << 20, + SessionId::from_bytes([1; 16]), + cellule_ltx::Host::default().with_dirty_slots(dirty.clone()), + ) + .unwrap(); + runtime.install_node_lease(f.lease.clone()).unwrap(); + runtime + .install_node_durability(ApplicationId::from_bytes([9; 16]), durability.clone()) + .unwrap(); + durability + .start_bundle_publication(authority.clone()) + .unwrap(); + let responses = Arc::new(Responses::default()); + runtime.install_telemetry(responses.clone()).unwrap(); + let mut cells = Vec::new(); + for byte in [4, 5] { + let target = CellTarget::new( + TenantId::from_bytes([1; 16]), + ApplicationId::from_bytes([9; 16]), + NamespaceId::from_bytes([13; 16]), + &[byte], + ) + .unwrap(); + let catalog = CellCatalog::new(f.layout.clone(), target.tenant()); + let proof = catalog + .provision( + CatalogEntry::new( + &target, + CatalogRole::Application, + Digest::from_bytes([12; 32]), + 1, + ) + .unwrap(), + ) + .await + .unwrap(); + let cell_authority = CellAuthority::new(f.layout.clone()); + let control = cell_authority + .create_initial( + &proof, + IncarnationId::from_bytes([byte; 16]), + Owner { + session: SessionId::from_bytes([1; 16]), + endpoint: "https://bundle.internal:8081".into(), + }, + ) + .await + .unwrap(); + let replica = CellReplica::new( + f.layout.clone(), + *target.cell_id().as_bytes(), + [byte; 16], + Limits::default(), + ) + .unwrap(); + let handle = runtime + .bootstrap( + proof, + replica.clone(), + cell_authority.clone(), + control, + f.scratch.path().join(format!("managed-{byte}.sqlite")), + |tx| { + tx.execute_batch( + "CREATE TABLE counter(value INTEGER); INSERT INTO counter VALUES(0)", + )?; + Ok(()) + }, + ) + .await + .unwrap(); + cells.push((target, cell_authority, replica, handle)); + } + // Ordinary per-Cell roots are unavailable. The producer must select exact + // origin coverage to grant any command ACK, read or retry visibility. + let preparation = dirty.acquire_owned().await.unwrap(); + let digest = Digest::from_bytes([9; 32]); + for byte in 1_u8..=20 { + let handle = &cells[usize::from(byte % 2)].3; + let identity = MutationIdentity { + request_id: RequestId::from_bytes([byte; 16]), + issued_at_ms: 10, + expires_at_ms: 10_000, + }; + let result = tokio::time::timeout( + std::time::Duration::from_secs(5), + handle.execute(identity, digest, 20, 1024, 1024, |tx| { + tx.execute("UPDATE counter SET value=value+1", [])?; + Ok(HandlerOutcome::Success(b"managed".to_vec())) + }), + ) + .await + .unwrap() + .unwrap(); + assert_eq!(result.commit_sequence(), u64::from(byte).div_ceil(2)); + assert_eq!( + handle + .execute(identity, digest, 21, 1024, 1024, |_| panic!( + "proved retry must not execute again" + )) + .await + .unwrap(), + result + ); + } + for (_, _, _, handle) in &cells { + assert_eq!( + handle + .query(1024, 1024, |connection| Ok(connection + .query_row("SELECT value FROM counter", [], |row| row.get::<_, i64>(0))? + .to_le_bytes() + .to_vec())) + .await + .unwrap(), + 10_i64.to_le_bytes() + ); + } + assert_eq!( + responses + .0 + .lock() + .unwrap() + .iter() + .filter(|source| **source == CommandResponseSource::Bundle) + .count(), + 20 + ); + let shutdown = runtime.shutdown(); + tokio::pin!(shutdown); + assert!( + tokio::time::timeout(std::time::Duration::from_millis(50), &mut shutdown) + .await + .is_err() + ); + drop(preparation); + peers.released.store(true, Ordering::Release); + peers.changed.notify_waiters(); + tokio::time::timeout(std::time::Duration::from_secs(10), &mut shutdown) + .await + .unwrap() + .unwrap(); + for (target, cell_authority, replica, _) in cells { + let control = cell_authority + .load(target.cell_id()) + .await + .unwrap() + .unwrap(); + assert_eq!(control.value().state, ControlState::Idle); + assert!(control.value().bundle_binding.is_none()); + let root = control.value().ltx_root().unwrap(); + assert_eq!(root.commit_sequence, 10); + let path = f + .scratch + .path() + .join(format!("cold-{}.sqlite", root.incarnation[0])); + replica + .open_root(&root) + .await + .unwrap() + .restore(&path) + .await + .unwrap(); + let cold = cellule_ltx::rusqlite::Connection::open(path).unwrap(); + assert_eq!( + cold.query_row("SELECT value FROM counter", [], |row| row.get::<_, i64>(0)) + .unwrap(), + 10 + ); + assert_eq!( + cold.query_row( + "SELECT COUNT(*) FROM sys_requests WHERE result=?1", + [b"managed".as_slice()], + |row| row.get::<_, i64>(0) + ) + .unwrap(), + 10 + ); + } + assert_eq!(durability.progress().unwrap().issued_through, 20); + assert_eq!( + pool.resource_ledger() + .snapshot() + .unwrap() + .used + .retained_bytes(), + 0 + ); +} diff --git a/crates/cellule-runtime/src/node/bundle/tests/mod.rs b/crates/cellule-runtime/src/node/bundle/tests/mod.rs index 8ede9952..a262596a 100644 --- a/crates/cellule-runtime/src/node/bundle/tests/mod.rs +++ b/crates/cellule-runtime/src/node/bundle/tests/mod.rs @@ -20,6 +20,7 @@ mod coverage; mod faults; mod index; mod lifecycle; +mod managed; mod ranges; mod receipts; mod recovery; diff --git a/crates/cellule-runtime/src/node/bundle/tests/receipts.rs b/crates/cellule-runtime/src/node/bundle/tests/receipts.rs index 6c63efea..16868f55 100644 --- a/crates/cellule-runtime/src/node/bundle/tests/receipts.rs +++ b/crates/cellule-runtime/src/node/bundle/tests/receipts.rs @@ -157,6 +157,30 @@ async fn selection_receipts_require_complete_cohort_and_admission_and_share_one_ assert!(Arc::ptr_eq(&first, &second)); assert!(first.proof.contains_assignment(&submitted[0].assignment)); assert!(second.proof.contains_assignment(&submitted[1].assignment)); + let prefix = durability + .capture_prefix(Arc::clone(&first), &submitted[0].assignment) + .unwrap(); + assert_eq!(prefix.proof.commit_sequence(), 2); + assert_eq!(prefix.proof.position(), cuts_by_commit[0].position); + assert_eq!( + prefix.proof.locator_count(), + cuts_by_commit[0].segments.len() + ); + assert!(prefix.proof.contains_assignment(&submitted[0].assignment)); + assert!(!prefix.proof.contains_assignment(&submitted[1].assignment)); + let first_root = f + .publisher(&cell) + .materialize_bundle(&prefix.proof) + .await + .unwrap(); + assert_eq!(first_root.commit_sequence, 2); + cell.control = cell + .authority + .load(cell.control.value().cell) + .await + .unwrap() + .unwrap(); + drop(prefix); drop(first); drop(second); drop(publication); diff --git a/crates/cellule-runtime/src/node/durability/mod.rs b/crates/cellule-runtime/src/node/durability/mod.rs index d77c6733..30441639 100644 --- a/crates/cellule-runtime/src/node/durability/mod.rs +++ b/crates/cellule-runtime/src/node/durability/mod.rs @@ -17,6 +17,8 @@ use crate::{Error, Result}; mod object_coverage; use object_coverage::ObjectCoverage; +mod publication; +pub use publication::{BundleCheckpoint, NodeBundlePublicationAuthority}; /// Original node authority used to enroll and close the Cells of an installed /// shared publication feed. Implementations serialize these mutations and shared @@ -191,6 +193,7 @@ pub struct NodeDurability { closed: std::sync::atomic::AtomicBool, selection_resources: std::sync::OnceLock, bundle_authority: std::sync::OnceLock>, + publisher: std::sync::OnceLock, } impl NodeDurability { @@ -239,6 +242,7 @@ impl NodeDurability { closed: std::sync::atomic::AtomicBool::new(false), selection_resources: std::sync::OnceLock::new(), bundle_authority: std::sync::OnceLock::new(), + publisher: std::sync::OnceLock::new(), } } @@ -305,6 +309,79 @@ impl NodeDurability { Ok(feed) } + /// Starts and retains the sole bounded node bundle producer. Install this + /// after the runtime resource ledger and before any Cell/native issuance. + /// Selection and checkpoints use the same original binding authority. + pub fn start_bundle_publication( + self: &Arc, + authority: Arc, + ) -> Result<()> { + if self.selection_resources.get().is_none() { + return Err(Error::Node( + "bundle publication has no installed runtime resource ledger", + )); + } + // Fallible construction precedes installation of the irreversible + // original feed. Rejected startup leaves the ordinary lane untouched. + let runtime = tokio::runtime::Handle::try_current().map_err(Error::RuntimeStart)?; + let working = publication::Publisher::reserve_working(self)?; + let original: Arc = authority.clone(); + let feed = self.enable_bundle_publication(original)?; + let publisher = publication::Publisher::start(self, authority, feed, working, runtime); + self.publisher + .set(publisher) + .map_err(|_| Error::Node("bundle producer already installed")) + } + + pub(crate) fn capture_prefix( + &self, + selected: Arc, + assignment: &crate::node::log::AssignedCommitRange, + ) -> Result> { + selected.proof.check_live_assignment(assignment)?; + let (_, commit, position) = assignment.endpoint(); + if selected.proof.commit_sequence() == commit && selected.proof.position() == position { + return Ok(selected); + } + let resources = self + .selection_resources + .get() + .ok_or(Error::PendingPublication)?; + // Reserve before cloning locator metadata; the original cohort remains + // separately charged until its other original consumers release it. + let memory = resources.try_reserve( + crate::fleet::resource::ResourceCost::zero() + .with_retained_bytes(selected.proof.retained_metadata_bytes()?), + )?; + let proof = selected.proof.original_capture_prefix(assignment)?; + Ok(Arc::new(crate::node::log_shipper::SelectedBundle { + proof, + _memory: memory, + })) + } + + pub(crate) async fn checkpoint_materialized( + &self, + authority: crate::control::authority::CellAuthority, + root: cellule_ltx::RootRef, + selected: Arc, + ) -> Result<()> { + if let Some(publisher) = self.publisher.get() { + publisher + .checkpoint(BundleCheckpoint { + authority, + root, + selected, + }) + .await?; + } + Ok(()) + } + + pub(crate) fn managed_bundle_publication(&self) -> bool { + self.publisher.get().is_some() + } + pub(crate) async fn bind_bundle_cell( &self, authority: &crate::control::authority::CellAuthority, @@ -358,6 +435,11 @@ impl NodeDurability { .ltx_root() .ok_or(Error::PendingPublication)?, )?; + // Never hold the provider's heartbeat/CAS mutex while waiting for the + // ordered producer: selecting the complete prior Fleet suffix needs it. + if let Some(publisher) = self.publisher.get() { + publisher.wait_through(issued.last_node_sequence()).await?; + } bundle.close(authority, observed, issued).await?; self.node_lease.check() } @@ -629,7 +711,13 @@ impl NodeDurability { } return Ok(proof); } - self.shipper.shutdown().await?; + let shipping = self.shipper.shutdown().await; + if let Some(publisher) = self.publisher.get() { + // Join both tasks even when fencing stopped the shipper. Preserve + // the producer's original cause rather than its secondary fence. + publisher.join().await?; + } + shipping?; self.object_coverage .flush(&self.gate, self.authority.as_ref(), &self.node_lease, &[]) .await?; diff --git a/crates/cellule-runtime/src/node/durability/publication/mod.rs b/crates/cellule-runtime/src/node/durability/publication/mod.rs new file mode 100644 index 00000000..af4ea129 --- /dev/null +++ b/crates/cellule-runtime/src/node/durability/publication/mod.rs @@ -0,0 +1,335 @@ +//! One bounded producer, exact root checkpoint queue, and joined epoch drain. +use super::*; +use crate::node::bundle::BundleCoverageProof; +use crate::node::log_shipper::{AssignedCapture, NodePublicationFeed, SelectedBundle}; +use std::sync::{Mutex as StdMutex, Weak}; +use tokio::sync::{mpsc, oneshot, watch}; + +const MAX_CAPTURES: usize = 64; +const MAX_FRAMES: usize = 64; +const MAX_NATIVE_BYTES: usize = crate::node::bundle::MAX_BUNDLE_BYTES as usize; +const MAX_CHECKPOINTS: usize = 512; +const ASSEMBLY: std::time::Duration = std::time::Duration::from_millis(1); + +/// Exact original selected metadata retained after its materialized root CAS. +/// Construction is restricted to the canonical actor publisher. +pub struct BundleCheckpoint { + pub(super) authority: crate::control::authority::CellAuthority, + pub(super) root: cellule_ltx::RootRef, + pub(super) selected: Arc, +} + +impl BundleCheckpoint { + /// Original Cell authority, including its application and origin scope. + pub fn authority(&self) -> &crate::control::authority::CellAuthority { + &self.authority + } + /// Exact complete-capture endpoint already materialized by the actor. + pub const fn root(&self) -> cellule_ltx::RootRef { + self.root + } + /// Original dependency-verified selection proof for this exact endpoint. + pub fn proof(&self) -> &BundleCoverageProof { + &self.selected.proof + } +} + +/// Serialized node origin operations for the runtime-owned publication task. +/// Implement with the same original state/heartbeat mutex as binding and close. +pub trait NodeBundlePublicationAuthority: NodeBundleAuthority { + /// Selects the complete ordered cohort with its native coverage in one CAS. + /// Return original live proofs only after dependency verification and CAS. + fn select<'a>( + &'a self, + captures: &'a [AssignedCapture], + lease: &'a NodeLeaseGuard, + ) -> BoxFuture<'a, Result>>; + /// Checkpoints exact materialized roots together. An obsolete notification + /// may be skipped only after observing a newer original materialized root; + /// the newer actor notification remains an independent joined obligation. + fn checkpoint<'a>(&'a self, checkpoints: &'a [BundleCheckpoint]) -> BoxFuture<'a, Result<()>>; +} + +#[derive(Clone, Default)] +struct Progress { + through: u64, + terminal: Option>>, +} + +type PublicationResult = std::result::Result<(), Arc>; +type PublicationTask = tokio::task::JoinHandle; + +pub(super) struct Publisher { + checkpoints: StdMutex>>, + progress: watch::Receiver, + worker: StdMutex>, + lease: NodeLeaseGuard, +} + +struct CheckpointRequest { + checkpoint: BundleCheckpoint, + completed: oneshot::Sender>>, +} + +impl Publisher { + pub(super) fn reserve_working( + durability: &NodeDurability, + ) -> Result { + let resources = durability + .selection_resources + .get() + .ok_or(Error::PendingPublication)?; + resources.try_reserve( + crate::fleet::resource::ResourceCost::zero() + .with_retained_bytes((4 * crate::node::bundle::MAX_BUNDLE_BYTES) as usize), + ) + } + + pub(super) fn start( + durability: &Arc, + authority: Arc, + feed: NodePublicationFeed, + working: crate::fleet::resource::ResourceReservation, + runtime: tokio::runtime::Handle, + ) -> Self { + // Preparation is admitted before SQL can consume the remainder of the + // ledger. Bounded catalog decode/encode copies cannot starve their own + // selected-capture cleanup while all foreground memory is retained. + let (sender, receiver) = mpsc::channel(MAX_CHECKPOINTS); + let (progress, observed) = watch::channel(Progress::default()); + let lease = durability.node_lease.clone(); + let running_lease = lease.clone(); + let weak = Arc::downgrade(durability); + let worker = runtime.spawn(async move { + let _working = working; + let result = run(weak, authority, feed, receiver, &progress, &running_lease) + .await + .map_err(Arc::new); + progress.send_modify(|state| state.terminal = Some(result.clone())); + if result.is_err() { + running_lease.fence(); + } + result + }); + Self { + checkpoints: StdMutex::new(Some(sender)), + progress: observed, + worker: StdMutex::new(Some(worker)), + lease, + } + } + + pub(super) async fn checkpoint(&self, checkpoint: BundleCheckpoint) -> Result<()> { + self.lease.check()?; + if checkpoint.root.commit_sequence != checkpoint.proof().commit_sequence() + || checkpoint.root.position != checkpoint.proof().position() + { + return Err(Error::Node("bundle checkpoint endpoint differs")); + } + let sender = self + .checkpoints + .lock() + .map_err(|_| Error::Node("bundle checkpoint lock poisoned"))? + .clone() + .ok_or(Error::RuntimeClosed)?; + let (completed, completion) = oneshot::channel(); + tokio::select! { + result = sender.send(CheckpointRequest { checkpoint, completed }) => result.map_err(|_| Error::RuntimeClosed)?, + () = self.lease.wait_fenced() => return self.terminal_error(), + } + // Cell departure is allowed only after this original callback joins, + // not after enqueue. Otherwise closing/detaching its pin could overtake + // a still-running catalog checkpoint from the accepted root task. + tokio::select! { + result = completion => result.map_err(|_| Error::RuntimeClosed)?.map_err(Error::Shared), + () = self.lease.wait_fenced() => self.terminal_error(), + } + } + + fn terminal_error(&self) -> Result<()> { + match &self.progress.borrow().terminal { + Some(Err(error)) => Err(Error::Shared(Arc::clone(error))), + _ => Err(Error::Fenced), + } + } + + pub(super) async fn wait_through(&self, through: u64) -> Result<()> { + let mut progress = self.progress.clone(); + loop { + { + let state = progress.borrow_and_update(); + if let Some(Err(error)) = &state.terminal { + return Err(Error::Shared(Arc::clone(error))); + } + if state.through >= through { + return self.lease.check(); + } + if state.terminal.is_some() { + return Err(Error::Node("bundle producer ended before issued range")); + } + } + tokio::select! { + result = progress.changed() => result.map_err(|_| Error::RuntimeClosed)?, + () = self.lease.wait_fenced() => return self.terminal_error(), + } + } + } + + pub(super) async fn join(&self) -> Result<()> { + self.checkpoints + .lock() + .map_err(|_| Error::Node("bundle checkpoint lock poisoned"))? + .take(); + let worker = self + .worker + .lock() + .map_err(|_| Error::Node("bundle producer lock poisoned"))? + .take(); + if let Some(worker) = worker { + return worker + .await + .map_err(Error::FollowerWorkerJoin)? + .map_err(Error::Shared); + } + // Cancellation of a joining caller detaches the retained worker; it + // does not cancel accepted I/O or manufacture a successful later join. + let mut progress = self.progress.clone(); + loop { + if let Some(result) = progress.borrow_and_update().terminal.clone() { + return result.map_err(Error::Shared); + } + progress.changed().await.map_err(|_| Error::RuntimeClosed)?; + } + } +} + +async fn run( + durability: Weak, + authority: Arc, + mut feed: NodePublicationFeed, + mut checkpoints: mpsc::Receiver, + progress: &watch::Sender, + lease: &NodeLeaseGuard, +) -> Result<()> { + let mut carry = None; + let mut checkpoints_open = true; + loop { + lease.check()?; + // One checkpoint cohort gets a turn even when native carryover never + // empties. Then prefer already queued native work over another root + // cohort; idle producers can still join all outstanding checkpoints. + if let Ok(first) = checkpoints.try_recv() { + tokio::select! { + result = checkpoint_cohort(&authority, &mut checkpoints, first) => result?, + () = lease.wait_fenced() => return Err(Error::Fenced), + } + } + let capture = if carry.is_some() { + carry.take() + } else if let Some(capture) = feed.try_recv() { + Some(capture) + } else { + tokio::select! { + capture = feed.recv() => capture, + checkpoint = checkpoints.recv(), if checkpoints_open => { + if let Some(first) = checkpoint { + tokio::select! { + result = checkpoint_cohort(&authority, &mut checkpoints, first) => result?, + () = lease.wait_fenced() => return Err(Error::Fenced), + } + } + else { checkpoints_open = false; } + continue; + } + () = lease.wait_fenced() => return Err(Error::Fenced), + } + }; + let Some(first) = capture else { + break; + }; + let mut captures = vec![first]; + let mut frames = captures[0].frames().len(); + let mut bytes = capture_bytes(&captures[0])?; + if frames > MAX_FRAMES || bytes > MAX_NATIVE_BYTES { + return Err(Error::Capacity( + "complete capture exceeds native bundle bounds", + )); + } + let deadline = tokio::time::Instant::now() + ASSEMBLY; + while captures.len() < MAX_CAPTURES { + let next = tokio::select! { + capture = feed.recv() => capture, + _ = tokio::time::sleep_until(deadline) => break, + () = lease.wait_fenced() => return Err(Error::Fenced), + }; + let Some(next) = next else { + break; + }; + let next_bytes = capture_bytes(&next)?; + if frames + next.frames().len() > MAX_FRAMES || bytes + next_bytes > MAX_NATIVE_BYTES { + carry = Some(next); + break; + } + frames += next.frames().len(); + bytes += next_bytes; + captures.push(next); + } + let original = durability.upgrade().ok_or(Error::RuntimeClosed)?; + let proofs = tokio::select! { + proofs = authority.select(&captures, lease) => proofs?, + () = lease.wait_fenced() => return Err(Error::Fenced), + }; + let selected = original.confirm_selected_captures(&captures, proofs)?; + progress.send_modify(|state| state.through = selected.selected_through()); + drop(selected); + drop(captures); + drop(original); + } + checkpoints.close(); + while let Some(first) = checkpoints.recv().await { + tokio::select! { + result = checkpoint_cohort(&authority, &mut checkpoints, first) => result?, + () = lease.wait_fenced() => return Err(Error::Fenced), + } + } + Ok(()) +} + +fn capture_bytes(capture: &AssignedCapture) -> Result { + capture.frames().iter().try_fold(0_usize, |sum, frame| { + sum.checked_add(frame.encoded().len()) + .ok_or(Error::Capacity("native bundle bytes")) + }) +} + +async fn checkpoint_cohort( + authority: &Arc, + receiver: &mut mpsc::Receiver, + first: CheckpointRequest, +) -> Result<()> { + let mut completions = vec![first.completed]; + let mut cohort = vec![first.checkpoint]; + while completions.len() < MAX_CAPTURES { + let Ok(next) = receiver.try_recv() else { + break; + }; + completions.push(next.completed); + let next = next.checkpoint; + if let Some(index) = cohort + .iter() + .position(|c| c.proof().binding() == next.proof().binding()) + { + if cohort[index].root.commit_sequence >= next.root.commit_sequence { + return Err(Error::Node("bundle checkpoint order regressed")); + } + cohort[index] = next; + } else { + cohort.push(next); + } + } + let result = authority.checkpoint(&cohort).await.map_err(Arc::new); + for completion in completions { + let _ = completion.send(result.clone()); + } + result.map_err(Error::Shared) +} diff --git a/crates/cellule-runtime/src/node/log/mod.rs b/crates/cellule-runtime/src/node/log/mod.rs index 6428b8a7..7478d49b 100644 --- a/crates/cellule-runtime/src/node/log/mod.rs +++ b/crates/cellule-runtime/src/node/log/mod.rs @@ -65,6 +65,10 @@ pub struct AssignedCommitRange { capture_digest: [u8; 32], } impl AssignedCommitRange { + pub(crate) const fn endpoint(&self) -> (u64, u64, cellule_ltx::Position) { + (self.first_commit, self.commit, self.position) + } + pub(crate) const fn scope(&self) -> CellLogScope { self.scope } diff --git a/crates/cellule-runtime/src/node/log_shipper/publication/mod.rs b/crates/cellule-runtime/src/node/log_shipper/publication/mod.rs index a5b57092..8ae99fd4 100644 --- a/crates/cellule-runtime/src/node/log_shipper/publication/mod.rs +++ b/crates/cellule-runtime/src/node/log_shipper/publication/mod.rs @@ -114,6 +114,10 @@ pub struct NodePublicationFeed { } impl NodePublicationFeed { + pub(crate) fn try_recv(&mut self) -> Option { + self.receiver.try_recv().ok() + } + /// Receives the next complete capture, including accepted work after close. /// None means all producers closed and every queued capture was received. pub async fn recv(&mut self) -> Option { diff --git a/crates/cellule-runtime/src/publication/mod.rs b/crates/cellule-runtime/src/publication/mod.rs index 105da03b..f626b0b2 100644 --- a/crates/cellule-runtime/src/publication/mod.rs +++ b/crates/cellule-runtime/src/publication/mod.rs @@ -216,6 +216,10 @@ impl CellPublisher { &self.observed } + pub(crate) fn authority(&self) -> &CellAuthority { + &self.authority + } + pub(crate) fn resource_limits(&self) -> cellule_ltx::Limits { self.replica.limits() } @@ -1230,6 +1234,33 @@ impl VerifiedBundleCapture { } impl PendingDurability { + pub(crate) async fn selected_capture_prefix(&self) -> Result> { + let Some(mut capture) = self.selected_capture().await? else { + return Ok(None); + }; + capture.selected = self + .durability + .capture_prefix(capture.selected, &capture.assignment)?; + Ok(Some(capture)) + } + + pub(crate) async fn checkpoint_materialized( + &self, + authority: &CellAuthority, + root: cellule_ltx::RootRef, + ) -> Result<()> { + if !self.durability.managed_bundle_publication() { + return Ok(()); + } + let capture = self + .selected_capture_prefix() + .await? + .ok_or(Error::Node("materialized bundle lost original capture"))?; + self.durability + .checkpoint_materialized(authority.clone(), root, capture.selected) + .await + } + pub(crate) fn has_bundle_capture(&self) -> bool { self.capture.is_some() } @@ -1371,6 +1402,12 @@ impl CellDurabilitySubmitter { Ok(capture) => capture, Err(error) => { self.check_node_lease()?; + // A bound ordered lane cannot silently omit a logical capture + // and later select beyond it. Preserve the submission error; + // the actor fences this uncertain already-committed outcome. + if durability.managed_bundle_publication() { + return Err(error); + } // The commit still succeeds through object coverage, so this // event and its counter are the only way to observe that an // enrolled lane refused the captured commit. From d4b2785eb350dac2227ab741cfaa87f81b8af458 Mon Sep 17 00:00:00 2001 From: forhappy Date: Thu, 8 Oct 2026 13:44:31 -0700 Subject: [PATCH 037/102] Record managed-producer TPS regression and failed qualification --- .../docs/write-performance-design.md | 27 +++- docs/bundle-coverage-implementation.md | 24 +-- docs/pr67-managed-producer-measurement.md | 147 ++++++++++++++++++ docs/write-performance-delivery.md | 19 ++- 4 files changed, 198 insertions(+), 19 deletions(-) create mode 100644 docs/pr67-managed-producer-measurement.md diff --git a/crates/cellule-runtime/docs/write-performance-design.md b/crates/cellule-runtime/docs/write-performance-design.md index 41fc6d31..8d36ac0e 100644 --- a/crates/cellule-runtime/docs/write-performance-design.md +++ b/crates/cellule-runtime/docs/write-performance-design.md @@ -50,9 +50,21 @@ flowchart LR Uploaded bytes are not authority. An explicitly installed original publication feed can now deliver admitted exact bundle receipts through the actor's existing -command, worker and read/retry gate. The example application's ordinary path -still has no node bundle producer. Production enablement requires the remaining -scheduler, recovery, collection and qualification gates. Do not advance the follower reclamation frontier before +command, worker and read/retry gate. `NodeDurability::start_bundle_publication` +now retains one producer under the installed runtime ledger. It selects complete +cohorts of at most 64 captures, 64 frames and 4 MiB, with a 1-ms assembly window +and a 16-MiB working reservation. Startup admission precedes installation of the +irreversible feed. Selection and exact root checkpoints use the same original +binding/heartbeat authority; 512 checkpoint requests are bounded and their +callbacks join before Cell closure. Fair turns alternate queued native work and +checkpoint cohorts. The Fleet SQL example installs this producer; the current +Bucket-only performance adapter bypasses it. + +The first end-to-end Fleet diagnostic of this connection failed throughput, +availability and drain. It is experimental, not performance qualification. +Per-Cell materializers, dense scheduling, complete failed-owner orchestration, +large-capture fallback, retryable producer failures and collection remain open. +Do not advance the follower reclamation frontier before failed-owner recovery understands the selected bundle prefix. ## Locator density and checkpoint cost @@ -133,8 +145,13 @@ normal root lineage, exact Cell CAS and joined drain complete. A later unproven suffix remains hidden. Origin materialization preserves the captured due time and retries storage errors within the existing publication grace. The bounded selection opportunity is 100 ms for installed feeds; absent or later coverage -uses the original root path. This is not the fair node materializer scheduler -or the 215-command checkpoint density target. +uses the original root path. A later cohort's proof can now be narrowed only by +the original complete-capture assignment, including its exact descriptors and +body digests. Materialized roots join the managed checkpoint callback before +releasing their publisher; Cell close waits for the complete original issued +prefix, including prior Fleet ACKs. The producer's task and failure cause also +join at epoch shutdown. This provides fair selection/checkpoint turns, not a +fair node materializer scheduler or the 215-command checkpoint density target. The live root fallback now avoids a node authority mutation when an exact Cell root covers only a sparse range beyond an unpublished native gap. That root diff --git a/docs/bundle-coverage-implementation.md b/docs/bundle-coverage-implementation.md index 8f2bf297..89183cfe 100644 --- a/docs/bundle-coverage-implementation.md +++ b/docs/bundle-coverage-implementation.md @@ -2,9 +2,12 @@ The connected protocol APIs now implement shared selection, independently awaitable root materialization and complete live-writer closure. An explicitly -installed original feed can now provide admitted receipts to the actor; -the example application's ordinary path still has no bundle producer. The prior performance regression -and failed qualification remain the baseline. This slice establishes ordering +installed original feed can now provide admitted receipts to the actor. +`NodeDurability::start_bundle_publication` retains a bounded producer and joins +exact checkpoint callbacks under the original authority. The Fleet SQL example +installs it; Bucket-only performance wiring bypasses it. Its first +[measured connection](pr67-managed-producer-measurement.md) regressed to 93.57 +Fleet writes/s from 525.67 and failed availability/drain. This slice establishes ordering and reconstruction evidence. The [fresh application-path benchmark](pr67-performance-reevaluation.md) measures `7fc0793`; it does not exercise bundle-based responses or establish write parity. The [WAL NORMAL comparison](pr67-normal-wal-reevaluation.md) @@ -88,13 +91,15 @@ fencing may still select that exact original base, but cannot reopen its closed catalog; departure then uses the ordinary closed-binding guard. Native frames issued before activation fail recovery before attachment or terminal selection, retaining the unresolved obligation. Admitted host scheduling and full lifecycle -qualification remain unfinished. Ordinary bundle responses remain disabled. +qualification remain unfinished. Bundle responses require the original installed +feed and admitted proof; the new connection is experimental and unqualified. The original SQL/capture/submission jobs must join before `close_cell_issuance`. Its ordered gate prevents late assignment from consuming a node sequence. The legacy identity-free `DurabilityGate::issue` cannot produce a per-Cell closure: -using it makes that closure fail closed. Automatic actor joining and scheduling -are still required before enabling this path for application commands. +using it makes that closure fail closed. Managed close now waits for the complete +issued producer prefix and joins exact checkpoint callbacks before Cell departure. +Failed-actor closure and dense materializer scheduling remain unqualified. ```mermaid sequenceDiagram @@ -374,9 +379,10 @@ reconstruction tests, not TPS qualification. Remaining work before qualified production enablement: -1. Install the canonical bounded node bundle producer and schedule fair admitted - materializer cohorts, coalescing proven debt rather than creating one root - per command. Preserve the integrated exact capture/visibility gate. +1. Build on the installed bounded producer: fix measured verification overhead + and failed-actor closure, then schedule fair admitted materializer cohorts, + coalescing proven debt rather than creating one root per command. Connect + Bucket-only publication and preserve exact capture/visibility gates. 2. Build on the authenticated index: bound admitted maintenance inventory, increase checkpoint density with retained-byte accounting, and measure the complete materialization/checkpoint/collection cost. diff --git a/docs/pr67-managed-producer-measurement.md b/docs/pr67-managed-producer-measurement.md new file mode 100644 index 00000000..4346495e --- /dev/null +++ b/docs/pr67-managed-producer-measurement.md @@ -0,0 +1,147 @@ +# Managed producer: measured write regression + +**The new Fleet connection regressed. Parity is not achieved.** It completed +93.57 successful writes/s versus 525.67 before the producer: **82.2% lower in +this pair**. Successful scheduled p99 grew from 155.13 to 3,501.22 ms. Warm +availability and joined drain failed. This experimental implementation remains +in draft PR #67; functional tests do not establish production readiness. + +Bucket current code completed 227.78/s versus 246.20/s: 7.5% lower in this +single pair. The Bucket-only fixture bypasses node durability and its producer, +so this difference cannot be attributed to the new shared producer. + +## Workload and identity + +Measured runtime commit: `2dc19172ce71d1f7b051705d70f005a70a47e554`. +Before-producer baseline: `3e436013a7e7500694bfdfe5600fe4a69adb1988`. +Celld: `f2bf648663a610eefde71f3547ad61e9b896b1f0` (pinned image). +Every exported candidate file was compared with the committed Git tree; +contents match, resolving original symlinks as the build export does. The build +manifest recorded the preceding HEAD because export preceded the code commit. +Its immutable source and binary hashes were preserved rather than rewritten. + +Each point used 1,000 uniform Cells, 96-byte SQL INSERT plus SELECT, a two-hour +request/result ledger, 128 clients and 128 queue slots. Warmup was 30 seconds; +the measured window was 60 seconds, one repetition. Fleet offered 15,000 writes/s +with two followers; Bucket offered 2,000/s. Read-only and mixed capacity were +not measured. This is SQL application parity, not the laptop's bounded KV load. + +All arms used the same Docker VM with 8 CPUs and **8 GiB total shared RAM**. +Owner/follower container ceilings were 8 CPUs/16 GiB with 4-GiB tmpfs; the client +ceiling was 4 CPUs/4 GiB. RustFS had 2 CPUs and an 8-GiB memory/swap ceiling, +changed from the canonical diagnostic runner's 2 GiB identically for every arm. +The loaded runner and this adaptation are hashed. No delivery, recovery or +qualification gate changed. Each case retained a fresh Linux Docker volume. +These ceilings exceed the shared VM's actual resources; another workstation VM +was present. This does not qualify a dedicated 8-CPU/16-GiB serving node. + +The first celld Fleet attempt was interrupted when its processes and VM +stopped, with no completed case summary. Its files and container states are +retained separately and excluded below. Both celld cases were rerun with fresh +prefixes. The comparator confirmed identical host/resource contracts, client +and auditor binaries, fixtures, pinned images and loaded runner across arms. +One serial overloaded pair cannot establish repeatability or isolated causality. + +## Reconciled windows + +TPS counts successful logical commands completed inside the window. Successful +p99 is nearest-rank replay of request journals, including trailing successful +completions; scheduled latency begins at offered arrival. Errors and drops are +excluded from these successful percentiles and remain explicit failures. + +| Mode / system | Successful writes/s | Successful scheduled p99 ms | Errors | Queue drops | +| --- | ---: | ---: | ---: | ---: | +| Fleet / before producer | 525.67 | 155.13 | 736,684 | 131,776 | +| Fleet / new producer | 93.57 | 3,501.22 | 76 | 894,057 | +| Fleet / celld | 4,238.78 | 66.73 | 9,334 | 636,199 | +| Bucket / before producer | 246.20 | 4,721.84 | 0 | 104,972 | +| Bucket / current code | 227.78 | 4,775.42 | 0 | 106,077 | +| Bucket / celld | 1,368.50 | 724.42 | 0 | 37,638 | + +Independent streaming replay reconciled every generated offer to a successful, +errored or dropped request, including trailing completions. ACK counts and +journal hashes also reconcile. **Every point fails delivery qualification**; +overloaded completion rates are not sustainable capacities. Celld Fleet also +changed posture and failed availability/recovery evidence; its high window rate +is not a qualified durable-throughput reference. + +## Recovery and failure evidence + +| Mode / system | ACK cohort | Warm | Bucket-only cold | Owner/fleet drain seconds | +| --- | ---: | --- | --- | ---: | +| Fleet / before producer | 44,760 | pass | pass | 7.76 | +| Fleet / new producer | 11,036 | fail: 173 errors | not completed | failed cleanup | +| Fleet / celld | 465,015 | fail: 427,303 errors | not completed | failed cleanup | +| Bucket / before producer | 24,924 | pass | pass | 7.27 | +| Bucket / current code | 23,925 | pass | pass | 10.00 | +| Bucket / celld | 129,955 | pass | pass | 1.78 | + +ACK cohorts include setup, warmup, steady and trailing successes. Audits GET +and retry every original acknowledged command, verifying stored output and +sequence. Current Fleet's warm errors were HTTP 503. Its owner failed the +120-second cleanup drain while reporting `PendingPublication`, then required +forced container stop. Cold audit was not reached. This establishes failure +of availability/drain, not a proved data-loss result. + +Celld Fleet's warm audit returned HTTP 500 errors; Docker recorded owner +`OOMKilled=true`, exit 137. Cold audit and successful drain were not reached. +Required after/cold provider observations are missing in both aborted cases; +missing evidence cannot pass. The interrupted attempt and failed rerun remain +retained. Successful tmpfs audits do not qualify physical power-loss durability. + +## Publication cost and next action + +| Fleet metric | Before producer | New producer | +| --- | ---: | ---: | +| Successful provider PUTs / completed command | 1.68 | 3.87 | +| GET and range calls / completed command | 0.48 | 24.30 | +| Materialized commands / selected Cell root | 2.35 | 1.09 | + +These are window-only storage API observations, excluding trailing work and +SDK-internal retries. The new producer was installed, but measured Fleet Bundle +response/proof counters did not advance: both remained four from setup/warmup. +Follower proof continued to win responses. The selector still performed work; +its 512-entry feed reserves space before native issuance, so slow publication +also backpressures Fleet commands. Memory was about 20 MiB at the initial sample +and 19 MiB at the final sample, including the 16-MiB producer reservation, below +the unchanged 64-MiB ledger ceiling at those sampled boundaries. + +The connected code retains and joins one bounded producer, fairly serves native +selection and exact checkpoint requests, narrows later-cohort proofs using +original complete assignments, joins checkpoint callbacks before Cell closure, +and waits for the complete issued prefix including prior Fleet ACKs. Three +managed-producer regressions test actor visibility/retry/cold results, rejected +startup admission, and preservation of the original producer failure cause. + +The next implementation must reduce repeated origin verification and sparse +materialization work, establish admitted dense materializer scheduling, and fix +failed-actor/overload closure. Increasing queues or dropping issued obligations +would not satisfy the contract. Bucket producer integration, large-capture +handling, retryable producer errors, complete failed-owner orchestration, safe +cross-Cell collection, and 2,000-Cell qualification remain open. The conditional +215-command checkpoint target is a component result, not the density above. + +## Verification and retained evidence + +The exact measured snapshot passed workspace checks and **1,954 tests**, with +zero failures and 38 documented ignored tests, local LTX tests, API docs, +boundary/layout, documentation and SQL/peer checks. Full workspace Clippy with +`-D warnings` passed on the declared Rust 1.97 minimum. Additional Rust 1.99 +Clippy failed on the pre-existing `AtomicU64::fetch_update` deprecation; that +failure is retained. No warning or performance gate was suppressed. Functional +verification does not erase the failed end-to-end result. + +Candidate source-manifest SHA-256: +`8c13c9ec32ba2dad6ca0857a6165ad661fdac49af3e4e2b9ccfd38a8f41498f1`. +Linux SQL release binary SHA-256: +`51d12763a1d81886d171b4c1592a86ec296f1eb50e57504d54206f8f933697b8`. +Raw journals, binaries, source snapshots, hashes, reports, failures and volumes +stay outside Git under +`/Volumes/Workspace/crabbuild-target/cellule-write-perf-8ad1`, label +`managed-producer-20261008`. Three paired five-minute repetitions, A/A variance, +stable debt, overload recovery, read guardrails and the dedicated serving-node +profile remain unqualified. See the unchanged +[runtime design](../crates/cellule-runtime/docs/write-performance-design.md). + +The external evidence index covers 1,854 retained files. Its SHA-256 is +`0475fdff305c8214de9b4f309a4dd74c4d152b9dc49384e1afd018633c726a4b`. diff --git a/docs/write-performance-delivery.md b/docs/write-performance-delivery.md index 2ecd4251..261dd0ed 100644 --- a/docs/write-performance-delivery.md +++ b/docs/write-performance-delivery.md @@ -6,7 +6,15 @@ comparison retains exact retry and cold-state audits. **Celld write parity has not been established.** The [proposal](write-performance-proposal.md) remains the acceptance contract; completing tests or a load run does not pass its gates. -The latest [selected-capture release measurement](pr67-selected-capture-release-measurement.md) +The latest [managed-producer measurement](pr67-managed-producer-measurement.md) +records runtime commit `2dc1917`: 93.57 Fleet writes/s versus 525.67 before +the producer, an 82.2% decrease in one matched overloaded pair. Successful p99, +warm availability and drain regressed. Bucket current code completed 227.78/s +versus 246.20/s; its fixture bypasses the producer. Celld completed 4,238.78 +Fleet/s with failed warm audit and OOM, and 1,368.50 Bucket/s with passing +ACK audits. Every point failed delivery qualification. PR #67 remains a draft. + +The earlier [selected-capture release measurement](pr67-selected-capture-release-measurement.md) records committed code at `3e43601`: 461.53 Fleet writes/s and 168.92 Bucket writes/s in one matched 60-second Docker diagnostic. Fleet was 5.9% higher than baseline; Bucket was 18.2% lower, with worse successful-write latency. All ACK @@ -75,7 +83,7 @@ collection paths. There is no legacy decoding or automatic migration. | M1 | Packs, inline leaves and bounded compaction spooling delivered | Ordinary append meets four PUTs; composed compaction needs five. Paired cold/sparse-read guardrail unverified | | M2 | Bounded file-backed shared publication coordinator implemented; exact scope, restore, cancellation, minimum-budget and dormant-sibling retention checks added | Latest three active-Fleet windows cost 5.229–5.433 PUTs/command; two fail the debt trend. Three per-Cell authority PUTs remain; M4 is required | | M3 | Signed 512-sequence/five-second grants, bounded local registry, lifecycle gate and signed HTTP fixture implemented | Full isolated checks and native lifecycle suite pass; latest three active-Fleet windows cost 0.0138 enrollment GETs/command. The 15K target diagnostic fails delivery and warm audit | -| M4 | [Connected protocol APIs](bundle-coverage-implementation.md), authenticated copy-on-write catalog shards, exact prefix/cohort checkpoints and streamed complete inventory | Actor response/read integration, admitted materializer scheduling, production checkpoint policy, failed-node issued-suffix recovery and bundle collection remain incomplete; bundle ACKs disabled | +| M4 | [Connected protocol APIs](bundle-coverage-implementation.md), actor command/read/retry and exact capture release, retained bounded producer, fair native/checkpoint turns and joined closure | New Fleet connection failed throughput, availability and drain; dense materializer scheduling, Bucket connection, failed-node orchestration, collection and qualification remain open | | M5 | Three paired low-rate Fleet repetitions and target diagnostics with exact ACK audits delivered | Publication stability and target delivery fail; qualified capacity, read/failure/overload matrix and absolute/relative parity remain unverified | ## Shared publication checkpoint @@ -90,9 +98,10 @@ The prior evidence below measures the packed implementation, **not this shared coordinator**. Its improvement percentages must not be attributed to M2. New source identities, checks and performance results will be recorded separately. Signed append grants are implemented with a fresh issuance path and local -durable closure gates. Bundle coverage ACKs remain disabled: the proposed -Cell-binding catalog, atomic transfer, exact range recovery and collection -contracts still need production integration. +durable closure gates. Bundle ACKs now require the installed original producer +and admitted exact proof. Its Fleet connection remains experimental after the +measured regression; dense scheduling, full recovery orchestration, collection +and qualification still need production integration. The first isolated M2/M3 snapshot passed 1,857 workspace tests (38 documented tests ignored), 58 local LTX tests without replica features, all-target/all-feature From 5193c7c0b2bde5cc247b4e8cdab70f309d821e14 Mon Sep 17 00:00:00 2001 From: forhappy Date: Thu, 8 Oct 2026 13:55:51 -0700 Subject: [PATCH 038/102] Join root coverage overtaken by native bundle selection --- .../docs/write-performance-design.md | 6 ++ .../src/node/bundle/tests/coverage.rs | 84 +++++++++++++++++++ .../cellule-runtime/src/node/directory/log.rs | 17 +++- .../src/node/durability/mod.rs | 5 +- 4 files changed, 108 insertions(+), 4 deletions(-) diff --git a/crates/cellule-runtime/docs/write-performance-design.md b/crates/cellule-runtime/docs/write-performance-design.md index 8d36ac0e..9b270fe6 100644 --- a/crates/cellule-runtime/docs/write-performance-design.md +++ b/crates/cellule-runtime/docs/write-performance-design.md @@ -153,6 +153,12 @@ prefix, including prior Fleet ACKs. The producer's task and failure cause also join at epoch shutdown. This provides fair selection/checkpoint turns, not a fair node materializer scheduler or the 215-command checkpoint density target. +Shared selection can persist a higher native coverage frontier while an older +root completion waits for the original authority mutex. In the same validated +open epoch, `advance_log_coverage` acknowledges that already selected prefix +without another CAS. It preserves the higher frontier; each local root still +confirms only its own original tickets. Expired or closed epochs still fail. + The live root fallback now avoids a node authority mutation when an exact Cell root covers only a sparse range beyond an unpublished native gap. That root grants its own object proof under the original lease; it does not advance follower diff --git a/crates/cellule-runtime/src/node/bundle/tests/coverage.rs b/crates/cellule-runtime/src/node/bundle/tests/coverage.rs index d0ff9133..f0505f13 100644 --- a/crates/cellule-runtime/src/node/bundle/tests/coverage.rs +++ b/crates/cellule-runtime/src/node/bundle/tests/coverage.rs @@ -254,6 +254,90 @@ async fn selection_advances_native_coverage_with_one_cas_while_roots_lag() { } } +#[tokio::test] +async fn queued_object_root_confirmation_joins_already_selected_native_coverage() { + let mut f = Fixture::new().await; + enroll(&mut f).await; + let mut a = f.cell(4).await; + let mut b = f.cell(5).await; + let (_, mut frames, arange) = f.append(&mut a, 2); + let (_, bframes, brange) = f.append(&mut b, 2); + frames.extend(bframes); + let prepared = f + .directory + .prepare_node_bundle(&f.node, &frames, &[arange, brange], NOW) + .await + .unwrap(); + let (selected, proofs) = f + .directory + .select_node_bundle(&f.node, &prepared, &f.lease, Limits::default(), NOW) + .await + .unwrap(); + // The shared authority CAS can overtake a root flusher's preview. Preserve + // the actual interval before the original producer confirms its local gate. + assert_eq!(selected.advertisement().log().unwrap().tiered_through(), 2); + assert_eq!(f.gate.tiered_through(), 0); + let proof = proofs + .iter() + .find(|proof| proof.binding() == a.control.value().bundle_binding.unwrap()) + .unwrap(); + f.publisher(&a).materialize_bundle(proof).await.unwrap(); + let authority = Arc::new(super::actor::Authority { + directory: f.directory.clone(), + observed: tokio::sync::Mutex::new(selected.clone()), + }); + let transport: Arc = Arc::new(NoExtraIo); + let shipper = crate::node::log_shipper::NodeLogShipper::new( + f.gate.clone(), + transport.clone(), + Limits::default(), + ) + .unwrap(); + let durability = crate::node::durability::NodeDurability::new( + f.gate.clone(), + shipper, + authority.clone(), + transport, + f.lease.clone(), + ); + f.count.reset(); + assert_eq!( + durability + .prove_object(arange.ticket()) + .await + .unwrap() + .source(), + DurabilitySource::Object + ); + assert_eq!( + f.count.put_requests(), + 0, + "already selected coverage needs no CAS" + ); + let current = authority.observed.lock().await.clone(); + assert_eq!(current.advertisement(), selected.advertisement()); + assert_eq!(current.advertisement().log().unwrap().tiered_through(), 2); + assert_eq!( + f.gate.tiered_through(), + 1, + "a root confirms only its own ticket" + ); + assert!(!f.gate.objects_are_covered(&[brange.ticket()]).unwrap()); + assert_eq!(durability.confirm_bundle(&proofs).unwrap(), 2); + assert_eq!( + durability.prove(brange.ticket()).await.unwrap().source(), + DurabilitySource::Bundle + ); + assert_eq!(f.count.put_requests(), 0); + assert!( + f.directory + .advance_log_coverage(¤t, 1, current.advertisement().expires_at_ms()) + .await + .is_err(), + "already covered work cannot bypass original lease expiry" + ); +} + #[tokio::test] async fn cold_proofs_and_replacement_gates_cannot_authorize_local_confirmation() { let mut f = Fixture::new().await; diff --git a/crates/cellule-runtime/src/node/directory/log.rs b/crates/cellule-runtime/src/node/directory/log.rs index ee67a373..e9f865bf 100644 --- a/crates/cellule-runtime/src/node/directory/log.rs +++ b/crates/cellule-runtime/src/node/directory/log.rs @@ -165,6 +165,8 @@ impl NodeDirectory { } /// CAS-advances the largest contiguous node sequence covered by object roots. + /// An already selected prefix is a no-op in the same validated open epoch; + /// shared native selection may overtake a queued root confirmation. pub async fn advance_log_coverage( &self, observed: &VersionedNodeAdvertisement, @@ -172,12 +174,21 @@ impl NodeDirectory { now_ms: i64, ) -> Result { self.validate(&observed.advertisement, now_ms)?; - let log = observed + let current = observed .advertisement .log .as_ref() - .ok_or(Error::Node("node session has no enrolled log"))? - .advance_tiered(observed.advertisement.node, tiered_through)?; + .ok_or(Error::Node("node session has no enrolled log"))?; + // Check the epoch phase even for an older completion. Keeping its + // already persisted frontier grants no new coverage and cannot reopen + // recovery/retirement. The caller still checks its original live lease. + let log = current.advance_tiered( + observed.advertisement.node, + tiered_through.max(current.tiered_through()), + )?; + if tiered_through <= current.tiered_through() { + return Ok(observed.clone()); + } let mut next = observed.advertisement.clone(); next.generation = next .generation diff --git a/crates/cellule-runtime/src/node/durability/mod.rs b/crates/cellule-runtime/src/node/durability/mod.rs index 30441639..07c8e0e1 100644 --- a/crates/cellule-runtime/src/node/durability/mod.rs +++ b/crates/cellule-runtime/src/node/durability/mod.rs @@ -67,7 +67,10 @@ pub trait NodeLogAuthority: Send + Sync { /// Activates one log epoch for this session. fn activate<'a>(&'a self, log_epoch: u64) -> BoxFuture<'a, Result<()>>; - /// Advances the tiered coverage watermark for one log epoch. + /// Advances the tiered coverage watermark for one log epoch. A queued root + /// completion may be older than shared bundle selection; acknowledge an + /// already persisted prefix under the same original epoch/heartbeat mutex + /// without lowering the frontier or reopening a closed epoch. fn advance_coverage<'a>( &'a self, log_epoch: u64, From e40ecd6b6b2972ee91a52846737cbc12bd48f406 Mon Sep 17 00:00:00 2001 From: forhappy Date: Thu, 8 Oct 2026 14:12:43 -0700 Subject: [PATCH 039/102] Bound full-compaction future stack usage on the minimum Rust version --- crates/cellule-runtime/src/publication/mod.rs | 77 ++++++++++--------- 1 file changed, 42 insertions(+), 35 deletions(-) diff --git a/crates/cellule-runtime/src/publication/mod.rs b/crates/cellule-runtime/src/publication/mod.rs index f626b0b2..f2532209 100644 --- a/crates/cellule-runtime/src/publication/mod.rs +++ b/crates/cellule-runtime/src/publication/mod.rs @@ -846,44 +846,51 @@ impl CellPublisher { } } - async fn force_full_compaction( - &mut self, - base: &cellule_ltx::RootRef, - ) -> Result> { - let segment_count = self - .segment_count - .ok_or(Error::Control("Cell segment count is unavailable"))?; - if segment_count <= 1 { - return Ok(None); - } - tracing::debug!(segments = segment_count, "Cell LTX full compaction forced"); - let mut backoff = Backoff::default(); - let prepared = loop { - let replica = self.replica.clone(); - let scratch_directory = self.scratch_directory.clone(); - let attempt = replica.prepare_compaction(base, 0..segment_count, 9, &scratch_directory); - tokio::pin!(attempt); - let result = loop { - tokio::select! { - result = &mut attempt => break result, - _ = tokio::time::sleep_until(tokio::time::Instant::from_std(self.renew_at)) => { - self.renew().await?; + fn force_full_compaction<'a>( + &'a mut self, + base: &'a cellule_ltx::RootRef, + ) -> futures_util::future::BoxFuture<'a, Result>> { + // Keep full compaction out of each enclosing publication poll frame. + // On the minimum Rust version the nested inline chain overflowed a + // default Tokio worker stack. The sole publisher still owns and joins + // this future through the same native/storage admissions. + Box::pin(async move { + let segment_count = self + .segment_count + .ok_or(Error::Control("Cell segment count is unavailable"))?; + if segment_count <= 1 { + return Ok(None); + } + tracing::debug!(segments = segment_count, "Cell LTX full compaction forced"); + let mut backoff = Backoff::default(); + let prepared = loop { + let replica = self.replica.clone(); + let scratch_directory = self.scratch_directory.clone(); + let attempt = + replica.prepare_compaction(base, 0..segment_count, 9, &scratch_directory); + tokio::pin!(attempt); + let result = loop { + tokio::select! { + result = &mut attempt => break result, + _ = tokio::time::sleep_until(tokio::time::Instant::from_std(self.renew_at)) => { + self.renew().await?; + } } + }; + self.record_publication_cost(); + match result { + Ok(prepared) => break prepared, + Err(error) if retryable_ltx_error(&error) => { + backoff.wait(ltx_retry_hint(&error)).await; + } + Err(error) => return Err(error.into()), } }; - self.record_publication_cost(); - match result { - Ok(prepared) => break prepared, - Err(error) if retryable_ltx_error(&error) => { - backoff.wait(ltx_retry_hint(&error)).await; - } - Err(error) => return Err(error.into()), - } - }; - let next_due_ms = self.observed.value().next_due_ms; - let root = self.publish_prepared(&prepared, next_due_ms).await?; - self.segment_count = Some(prepared.verified().segment_count()); - Ok(Some(root)) + let next_due_ms = self.observed.value().next_due_ms; + let root = self.publish_prepared(&prepared, next_due_ms).await?; + self.segment_count = Some(prepared.verified().segment_count()); + Ok(Some(root)) + }) } async fn prepare_cuts( From df922e7b2c3ecc721e688fd0e613315ac38b7220 Mon Sep 17 00:00:00 2001 From: forhappy Date: Thu, 8 Oct 2026 14:45:02 -0700 Subject: [PATCH 040/102] Record measured recovery fix without claiming write throughput gain --- .../docs/write-performance-design.md | 7 +- docs/bundle-coverage-implementation.md | 10 +- docs/pr67-coverage-race-measurement.md | 154 ++++++++++++++++++ docs/pr67-managed-producer-measurement.md | 4 + docs/write-performance-delivery.md | 17 +- 5 files changed, 183 insertions(+), 9 deletions(-) create mode 100644 docs/pr67-coverage-race-measurement.md diff --git a/crates/cellule-runtime/docs/write-performance-design.md b/crates/cellule-runtime/docs/write-performance-design.md index 9b270fe6..72a71060 100644 --- a/crates/cellule-runtime/docs/write-performance-design.md +++ b/crates/cellule-runtime/docs/write-performance-design.md @@ -62,6 +62,9 @@ Bucket-only performance adapter bypasses it. The first end-to-end Fleet diagnostic of this connection failed throughput, availability and drain. It is experimental, not performance qualification. +The subsequent [coverage-race measurement](../../../docs/pr67-coverage-race-measurement.md) +at `e40ecd6` passes warm/cold ACK read/retry and joined drain, but completes +100.20 Fleet writes/s versus 106.05 before the fix. There is no measured speedup. Per-Cell materializers, dense scheduling, complete failed-owner orchestration, large-capture fallback, retryable producer failures and collection remain open. Do not advance the follower reclamation frontier before @@ -165,5 +168,5 @@ grants its own object proof under the original lease; it does not advance follow reclamation or permit rotation. Closing the gap still persists the complete new contiguous frontier before confirming it locally. Failed or cancelled advancing CAS work remains staged for retry and joined drain. This removes redundant -coordination work from the existing application path; bundle ACK integration -and the node capacity qualification remain open. +coordination work from the existing application path; Bucket producer connection, +dense materialization and the node capacity qualification remain open. diff --git a/docs/bundle-coverage-implementation.md b/docs/bundle-coverage-implementation.md index 89183cfe..21f12ac0 100644 --- a/docs/bundle-coverage-implementation.md +++ b/docs/bundle-coverage-implementation.md @@ -7,13 +7,17 @@ installed original feed can now provide admitted receipts to the actor. exact checkpoint callbacks under the original authority. The Fleet SQL example installs it; Bucket-only performance wiring bypasses it. Its first [measured connection](pr67-managed-producer-measurement.md) regressed to 93.57 -Fleet writes/s from 525.67 and failed availability/drain. This slice establishes ordering -and reconstruction evidence. The [fresh application-path benchmark](pr67-performance-reevaluation.md) +Fleet writes/s from 525.67 and failed availability/drain. The subsequent +[coverage-race measurement](pr67-coverage-race-measurement.md) at `e40ecd6` +passes all-ACK warm/cold audit and joined drain, but completes 100.20 Fleet +writes/s versus 106.05 before the fix: no demonstrated throughput gain. +This slice establishes ordering and reconstruction evidence. +The [earlier application-path benchmark](pr67-performance-reevaluation.md) measures `7fc0793`; it does not exercise bundle-based responses or establish write parity. The [WAL NORMAL comparison](pr67-normal-wal-reevaluation.md) separately records seven completed diagnostic cases and an interrupted matrix. -The [latest sparse-root coverage measurement](pr67-sparse-root-coverage-measurement.md) +The [earlier sparse-root coverage measurement](pr67-sparse-root-coverage-measurement.md) exercises the ordinary application path at `b185672`: 378.88 completed writes/s versus 282.43 before the change in one five-minute Fleet-configured pair. Successful-write p99 improved, median latency worsened, and audit, drain and diff --git a/docs/pr67-coverage-race-measurement.md b/docs/pr67-coverage-race-measurement.md new file mode 100644 index 00000000..c8cbf359 --- /dev/null +++ b/docs/pr67-coverage-race-measurement.md @@ -0,0 +1,154 @@ +# Coverage race fix: measured recovery improvement, no throughput gain + +**Latest measured code: 100.20 Fleet writes/s and 271.63 Bucket writes/s.** +Compared with the immediately preceding code, Fleet completed 5.5% fewer writes +and Bucket 5.3% fewer in these single overloaded pairs. Successful scheduled p99 +increased 9.3% and 14.2%, respectively. There is no demonstrated throughput or +latency improvement. The coverage fix restored Fleet warm availability and +joined drain in this run; all acknowledged commands then passed cold read/retry +verification. **Parity is not achieved. PR #67 remains a draft.** + +## Changes and immutable identities + +Measured candidate: `e40ecd6b6b2972ee91a52846737cbc12bd48f406`. +Before-fix code: `2dc19172ce71d1f7b051705d70f005a70a47e554`. +Celld: `f2bf648663a610eefde71f3547ad61e9b896b1f0`, the same pinned image as the +[previous producer comparison](pr67-managed-producer-measurement.md). +The baseline reuses the exact previously verified release binary, with a fresh +case, prefix and provider volume. Its immutable build manifest records export +HEAD `7b986b0`; a complete Git-tree comparison established that the exported +contents equal committed `2dc1917`. That manifest was not rewritten. +The new candidate was exported and built from committed `e40ecd6`; its frozen +verification snapshot and release source manifest match that commit. + +| Fix | Verification | +| --- | --- | +| `5193c7c`: join root coverage overtaken by native bundle selection | The real original-authority regression failed before the fix with `node log object coverage regressed`, then passed. An older root acknowledges an already persisted prefix with zero PUTs; the higher frontier stays unchanged. Local root confirmation covers only its own tickets. Open-epoch, expiry, lease and complete-assignment checks remain mandatory. | +| `e40ecd6`: box the full-compaction future | The existing 65-command backlog test overflowed the default worker stack on Rust 1.97, including an exact pre-fix baseline reproduction. The same test passes after boxing the joined compaction transition, without increasing the stack or changing publication/compaction ordering. | + +These fixes add no persisted or wire format change. A separate warning-only +diagnostic captured 30 coverage-regression warnings; its TPS is excluded because +it used a debug subscriber and briefly overlapped an interrupted native build. +Compiler type-size probes are diagnostic evidence only. Neither diagnostic is +included in the six windows below. + +## Matched workload and limits + +All six cases ran sequentially with the same client/auditor, fixture hashes, +loaded runner, pinned images and Docker host/resource contract. Independent +comparison verified those identities. Each used 1,000 uniform Cells, 96-byte SQL +INSERT plus SELECT, a two-hour request/result ledger, 128 clients and 128 queue +slots, 30-second warmup and one 60-second measured window. Fleet offered 15,000 +writes/s with two followers; Bucket offered 2,000/s. This is the SQL application +profile, not bounded KV overwrite. Standalone read and mixed capacity were not +measured. + +The ARM64 Docker VM has **8 CPUs and 8 GiB total shared RAM**. Each serving node +has an 8-CPU/16-GiB container ceiling and 4-GiB tmpfs; the client ceiling is +4 CPUs/4 GiB. RustFS has 2 CPUs and an 8-GiB memory/swap ceiling, using the same +external diagnostic adaptation for all arms. The ceilings exceed actual shared +VM resources, and another workstation VM was present. Every case used a fresh +Linux provider volume. This does not qualify the dedicated 8-CPU/16-GiB node, +2,000-Cell population, physical-media durability or repeatability requirements. +No delivery, durability, recovery or qualification gate was weakened. + +## Reconciled windows + +TPS counts successful logical commands completed inside the measured window. +Successful p99 is nearest-rank replay of the request journals, including trailing +successes. Scheduled latency starts at offered arrival; request latency starts +at actual request issuance. Errors and drops are excluded from successful +percentiles and remain explicit failures. The unchanged delivery gate uses +all-attempt scheduled latency as well as delivery and recovery evidence. + +| Mode / system | Successful writes/s | Successful scheduled p99 ms | Successful request p99 ms | Errors | Queue drops | +| --- | ---: | ---: | ---: | ---: | ---: | +| Fleet / before fix | 106.05 | 3,507.00 | 2,379.51 | 76 | 893,306 | +| Fleet / latest | 100.20 | 3,832.17 | 2,298.88 | 0 | 893,732 | +| Fleet / celld | 4,202.05 | 179.78 | 41.64 | 13,715 | 634,002 | +| Bucket / before fix | 286.93 | 3,561.76 | 3,087.89 | 0 | 102,528 | +| Bucket / latest | 271.63 | 4,065.98 | 3,599.15 | 0 | 103,446 | +| Bucket / celld | 1,472.08 | 672.38 | 342.38 | 0 | 31,425 | + +Streaming replay reconciled every measured offer to a success, error or drop, +including trailing completions, and reconciled ACK cohorts and their hashes. +**Every point fails delivery qualification.** These overloaded completion rates +are not sustainable capacities. One pair cannot establish an attributable +performance regression or improvement. The previous producer regression remains +historical evidence; these results do not reverse it or establish parity. + +## Recovery and availability + +| Mode / system | ACK cohort | Warm read/retry | Bucket-only cold read/retry | Drain seconds | +| --- | ---: | --- | --- | ---: | +| Fleet / before fix | 11,855 | fail: 184 HTTP 503 errors | not reached | failed: owner wait exceeded 120 seconds | +| Fleet / latest | 11,197 | pass | pass | 25.40 | +| Fleet / celld | 468,386 | fail: 378,252 HTTP 500 errors | not reached | failed cleanup | +| Bucket / before fix | 26,858 | pass | pass | 5.51 | +| Bucket / latest | 28,009 | pass | pass | 6.29 | +| Bucket / celld | 134,965 | pass | pass | 3.58 | + +Cohorts include setup, warmup, steady and trailing successes. Passing audits +GET and retry every original ACK, checking stored outcomes and incarnation. +Both latest Cellule cases also passed the cold contract retry. Cold cases start +from empty local state after joined graceful drain; these are not failed-owner +recovery or owner-loss-before-materialization tests. + +Celld Fleet's owner has `OOMKilled: true`, exit 137, in retained Docker state. +Its high completed-write rate is not qualified durable capacity. Warm audit and +cleanup failures establish availability/recovery-evidence failures, not proven +data loss. Missing post-failure provider/cold observations remain failed evidence +in the reports rather than being filled with assumed values. + +## Bottleneck still measured + +Latest Fleet costs **3.65 successful PUTs and 24.03 GET/range attempts per +completed write**, versus 3.65 and 24.20 before the fix. Materialization density +is **1.12 commands/root**, versus 1.10 before. Mean capture is 0.29 ms, worker +round trip 3.24 ms and Fleet proof 6.00 ms; mean publication is 5,479.53 ms and +Fleet response 1,216.75 ms. These phase samples have different cohorts and cannot +be summed as a per-request critical path. + +The bundle response counter remains two across the steady window: it adds zero +steady Bundle ACKs. The original native epoch stays Fleet-active, but pending +native object sequences grow 309 → 486. Retained runtime bytes stay around +19 MiB under the 64-MiB ledger, including the producer's 16-MiB reservation. +These two boundaries do not establish stable debt over five minutes. + +Code inspection shows the publication feed applies backpressure before native +sequence issuance, and the same producer alternates selection and root +checkpoint work. Together with the measured I/O and sparse root density, this +identifies publication/verification work as the next optimization target. The +data does not support attributing the throughput gap to SQLite sync or capture +cost. Dense admitted materialization, reduced repeated verification, Bucket +producer integration, failed-owner complete-suffix recovery, collection and +full read/write/mixed qualification remain open. The original 215-command +checkpoint model and 2,000-Cell/10K-write/50K-read gates remain unchanged. + +## Verification and evidence + +The exact frozen candidate passed all eleven contributor routes on Rust 1.97: +format, workspace targets/features, workspace tests (**1,955 passed, zero failed, +38 documented ignored**), local LTX, Clippy with `-D warnings`, API docs, +boundaries, module layout, Rust fences, links and SQL/peer contracts. Linux +release builds used the same pinned Rust 1.98.1 image. The earlier stack failure +and the previously recorded Rust 1.99 Clippy deprecation remain retained; no +warning or stack setting was suppressed to pass. + +Raw journals, binaries, logs, source manifests and provider volume records stay +outside Git under `/Volumes/Workspace/crabbuild-target/cellule-write-perf-8ad1`. + +| Evidence | Identity | +| --- | --- | +| Latest source manifest | `83114274a5fe95e14ac9eedced32553c71ba38429578b58152080740a849d416` | +| Latest SQL binary | `0bd881641c72cbee662f22d1a0751823ff1a3c54c3115d73bebb76e8c5d00f22` | +| Baseline source manifest | `8c13c9ec32ba2dad6ca0857a6165ad661fdac49af3e4e2b9ccfd38a8f41498f1` | +| Baseline SQL binary | `51d12763a1d81886d171b4c1592a86ec296f1eb50e57504d54206f8f933697b8` | +| Identical client | `417f07b0424d27df75b1dca22666a7adb621db3cd11fdb1048eb9247a664e60b` | +| Identical auditor | `6595c24b0be217e181e8a9aa7de5cf478669ac341b37b76754fdcf0b3865b7b2` | +| Evidence index: 1,923 files | `fd5100792b8853d84211949e0863e518bb36b4eec5fddd3d8c5f00b277b54ab2` | + +The index is `coverage-race-stack-20261008-evidence-index.json`. Independent +replay is `coverage-race-stack-20261008-reconciled.json`; paired matrices and +comparisons are `coverage-race-stack-20261008-{fleet,bucket}-{matrix,comparison}.json`. +They preserve failed cases and explicitly report `qualification_pass: false`. diff --git a/docs/pr67-managed-producer-measurement.md b/docs/pr67-managed-producer-measurement.md index 4346495e..34d6c29d 100644 --- a/docs/pr67-managed-producer-measurement.md +++ b/docs/pr67-managed-producer-measurement.md @@ -1,5 +1,9 @@ # Managed producer: measured write regression +Historical measurement. The subsequent +[coverage-race fix and fresh comparison](pr67-coverage-race-measurement.md) +restore Fleet warm/cold availability and drain, but demonstrate no TPS gain. + **The new Fleet connection regressed. Parity is not achieved.** It completed 93.57 successful writes/s versus 525.67 before the producer: **82.2% lower in this pair**. Successful scheduled p99 grew from 155.13 to 3,501.22 ms. Warm diff --git a/docs/write-performance-delivery.md b/docs/write-performance-delivery.md index 261dd0ed..517a477e 100644 --- a/docs/write-performance-delivery.md +++ b/docs/write-performance-delivery.md @@ -6,7 +6,16 @@ comparison retains exact retry and cold-state audits. **Celld write parity has not been established.** The [proposal](write-performance-proposal.md) remains the acceptance contract; completing tests or a load run does not pass its gates. -The latest [managed-producer measurement](pr67-managed-producer-measurement.md) +The latest [coverage-race measurement](pr67-coverage-race-measurement.md) +records runtime commit `e40ecd6`: 100.20 Fleet writes/s and 271.63 Bucket writes/s, +5.5% and 5.3% lower than the immediately preceding code in one fresh pair. +Successful scheduled p99 also worsened. Fleet warm availability, joined drain +and all-ACK cold read/retry now pass; there is no demonstrated throughput or +latency gain. Fresh celld completed 4,202.05 Fleet/s with failed warm audit and +OOM, and 1,472.08 Bucket/s with passing ACK audits. Every point fails delivery +qualification; PR #67 remains a draft. + +The earlier [managed-producer measurement](pr67-managed-producer-measurement.md) records runtime commit `2dc1917`: 93.57 Fleet writes/s versus 525.67 before the producer, an 82.2% decrease in one matched overloaded pair. Successful p99, warm availability and drain regressed. Bucket current code completed 227.78/s @@ -81,9 +90,9 @@ collection paths. There is no legacy decoding or automatic migration. | --- | --- | --- | | M0 | Measurement and comparison harness delivered | Three A/A capacity pairs unverified; storage API totals reconcile, but SDK-internal HTTP retries need provider telemetry | | M1 | Packs, inline leaves and bounded compaction spooling delivered | Ordinary append meets four PUTs; composed compaction needs five. Paired cold/sparse-read guardrail unverified | -| M2 | Bounded file-backed shared publication coordinator implemented; exact scope, restore, cancellation, minimum-budget and dormant-sibling retention checks added | Latest three active-Fleet windows cost 5.229–5.433 PUTs/command; two fail the debt trend. Three per-Cell authority PUTs remain; M4 is required | -| M3 | Signed 512-sequence/five-second grants, bounded local registry, lifecycle gate and signed HTTP fixture implemented | Full isolated checks and native lifecycle suite pass; latest three active-Fleet windows cost 0.0138 enrollment GETs/command. The 15K target diagnostic fails delivery and warm audit | -| M4 | [Connected protocol APIs](bundle-coverage-implementation.md), actor command/read/retry and exact capture release, retained bounded producer, fair native/checkpoint turns and joined closure | New Fleet connection failed throughput, availability and drain; dense materializer scheduling, Bucket connection, failed-node orchestration, collection and qualification remain open | +| M2 | Bounded file-backed shared publication coordinator implemented; exact scope, restore, cancellation, minimum-budget and dormant-sibling retention checks added | Earlier three active-Fleet windows cost 5.229–5.433 PUTs/command; two fail the debt trend. Per-Cell authority work remains; M4 is required | +| M3 | Signed 512-sequence/five-second grants, bounded local registry, lifecycle gate and signed HTTP fixture implemented | Full isolated checks and native lifecycle suite pass; earlier three active-Fleet windows cost 0.0138 enrollment GETs/command. The latest 15K diagnostic still fails delivery despite passing ACK audits | +| M4 | [Connected protocol APIs](bundle-coverage-implementation.md), actor command/read/retry and exact capture release, retained bounded producer, fair native/checkpoint turns, monotonic coverage joining and joined closure | Latest Fleet warm/cold ACK audit and drain pass; throughput/latency targets still fail. Dense materializer scheduling, Bucket connection, failed-node orchestration, collection and qualification remain open | | M5 | Three paired low-rate Fleet repetitions and target diagnostics with exact ACK audits delivered | Publication stability and target delivery fail; qualified capacity, read/failure/overload matrix and absolute/relative parity remain unverified | ## Shared publication checkpoint From 9d4e6328bceb2f88087698809c3b4167f94959ef Mon Sep 17 00:00:00 2001 From: forhappy Date: Thu, 8 Oct 2026 15:04:37 -0700 Subject: [PATCH 041/102] Verify fresh bundle cohorts from one bounded origin read --- .../docs/write-performance-design.md | 11 ++- .../src/node/bundle/index/io.rs | 75 ++++++++--------- .../src/node/bundle/index/tests/admission.rs | 2 +- crates/cellule-runtime/src/node/bundle/mod.rs | 7 +- .../cellule-runtime/src/node/bundle/origin.rs | 80 +++++++++++++++++++ .../cellule-runtime/src/node/bundle/proof.rs | 52 +++++++++--- .../src/node/bundle/selection.rs | 7 +- .../cellule-runtime/src/node/bundle/store.rs | 2 +- .../src/node/bundle/tests/index/cohort.rs | 73 +++++++++++++++++ .../src/node/durability/publication/mod.rs | 4 +- 10 files changed, 258 insertions(+), 55 deletions(-) create mode 100644 crates/cellule-runtime/src/node/bundle/origin.rs diff --git a/crates/cellule-runtime/docs/write-performance-design.md b/crates/cellule-runtime/docs/write-performance-design.md index 72a71060..ed8d110e 100644 --- a/crates/cellule-runtime/docs/write-performance-design.md +++ b/crates/cellule-runtime/docs/write-performance-design.md @@ -53,13 +53,22 @@ feed can now deliver admitted exact bundle receipts through the actor's existing command, worker and read/retry gate. `NodeDurability::start_bundle_publication` now retains one producer under the installed runtime ledger. It selects complete cohorts of at most 64 captures, 64 frames and 4 MiB, with a 1-ms assembly window -and a 16-MiB working reservation. Startup admission precedes installation of the +and a 20-MiB working reservation, including one bounded fresh origin read. +Startup admission precedes installation of the irreversible feed. Selection and exact root checkpoints use the same original binding/heartbeat authority; 512 checkpoint requests are bounded and their callbacks join before Cell closure. Fair turns alternate queued native work and checkpoint cohorts. The Fleet SQL example installs this producer; the current Bucket-only performance adapter bypasses it. +Each selection reads its complete new cohort object once from origin, compares +every byte with the proposal, then verifies header, shards, histories and native +frames from that operation's read. It retains no cross-operation availability +cache. Historical objects and every Cell base dependency still require origin +verification. Selection drops each checked native frame rather than retaining +reconstruction bodies. The additional 4-MiB buffer is charged before installing +the producer; workload retention and protocol bounds remain unchanged. + The first end-to-end Fleet diagnostic of this connection failed throughput, availability and drain. It is experimental, not performance qualification. The subsequent [coverage-race measurement](../../../docs/pr67-coverage-race-measurement.md) diff --git a/crates/cellule-runtime/src/node/bundle/index/io.rs b/crates/cellule-runtime/src/node/bundle/index/io.rs index 1a5da4d2..5d5e0067 100644 --- a/crates/cellule-runtime/src/node/bundle/index/io.rs +++ b/crates/cellule-runtime/src/node/bundle/index/io.rs @@ -5,13 +5,17 @@ async fn load_root( layout: &cellule_ltx::CellStorageLayout, session: SessionId, head: NodeBundleHead, + origin: Option<&super::super::origin::OriginBundle>, ) -> Result> { - let path = - layout.node_coverage_bundle_path(session.as_bytes(), head.epoch, head.digest.as_bytes()); - let header = layout - .store() - .range_get(&path, 0..HEADER_BYTES as u64) - .await?; + let header = super::super::origin::read_range( + layout, + session, + head.epoch, + head.digest, + 0..HEADER_BYTES as u64, + origin, + ) + .await?; if !matches!(header.get(..8), Some(magic) if magic == MAGIC || magic == DENSE_MAGIC) { return Ok(None); } @@ -38,22 +42,21 @@ async fn load_rows( root: &Root, shard: &Shard, id: u8, + origin: Option<&super::super::origin::OriginBundle>, ) -> Result<(Vec, BTreeMap<[u8; 32], history::History>)> { let extent = &shard.extent; let object = extent .object .ok_or(Error::Node("unresolved bundle catalog shard"))?; - let bytes = layout - .store() - .range_get( - &layout.node_coverage_bundle_path( - root.session.as_bytes(), - root.epoch, - object.as_bytes(), - ), - extent.offset..extent.offset + extent.bytes, - ) - .await?; + let bytes = super::super::origin::read_range( + layout, + root.session, + root.epoch, + object, + extent.offset..extent.offset + extent.bytes, + origin, + ) + .await?; let (mut leaf, mut histories) = decode_leaf_with_histories(bytes, shard, id, root)?; for locator in leaf .bindings @@ -79,7 +82,7 @@ pub(in crate::node::bundle) async fn load( head: NodeBundleHead, wanted: Option<&BTreeSet>, ) -> Result { - load_inner(layout, session, head, wanted, None).await + load_inner(layout, session, head, wanted, None, None).await } pub(in crate::node::bundle) async fn load_cells( @@ -87,12 +90,13 @@ pub(in crate::node::bundle) async fn load_cells( session: SessionId, head: NodeBundleHead, cells: &BTreeSet, + origin: Option<&super::super::origin::OriginBundle>, ) -> Result { let shards = cells .iter() .map(|(application, cell)| shard(application, cell)) .collect(); - load_inner(layout, session, head, Some(&shards), Some(cells)).await + load_inner(layout, session, head, Some(&shards), Some(cells), origin).await } async fn load_inner( @@ -101,8 +105,9 @@ async fn load_inner( head: NodeBundleHead, wanted: Option<&BTreeSet>, cells: Option<&BTreeSet>, + origin: Option<&super::super::origin::OriginBundle>, ) -> Result { - let Some(root) = load_root(layout, session, head).await? else { + let Some(root) = load_root(layout, session, head, origin).await? else { return super::super::store::load_legacy_catalog(layout, session, head).await; }; // An immutable shard can be individually bounded while a selected cohort @@ -127,7 +132,7 @@ async fn load_inner( continue; } let (rows, leaf_histories) = match &root.shards[usize::from(id)] { - Some(shard) => load_rows(layout, &root, shard, id).await?, + Some(shard) => load_rows(layout, &root, shard, id, origin).await?, None => (Vec::new(), BTreeMap::new()), }; for (pin, history) in leaf_histories { @@ -179,17 +184,15 @@ async fn load_inner( .extent .object .ok_or(Error::Node("unresolved bundle history"))?; - let bytes = layout - .store() - .range_get( - &layout.node_coverage_bundle_path( - session.as_bytes(), - head.epoch, - object.as_bytes(), - ), - history.extent.offset..history.extent.offset + history.extent.bytes, - ) - .await?; + let bytes = super::super::origin::read_range( + layout, + session, + head.epoch, + object, + history.extent.offset..history.extent.offset + history.extent.bytes, + origin, + ) + .await?; let mut locators = history::decode(&bytes, session, head.epoch, binding, history)?; for locator in &mut locators { if locator.object.is_none() { @@ -235,14 +238,14 @@ pub(in crate::node::bundle) async fn ensure_drained( session: SessionId, head: NodeBundleHead, ) -> Result<()> { - let Some(root) = load_root(layout, session, head).await? else { + let Some(root) = load_root(layout, session, head, None).await? else { let catalog = super::super::store::load_legacy_catalog(layout, session, head).await?; return check_closed(&catalog.bindings); }; let mut pins = std::collections::HashSet::new(); for (id, shard) in root.shards.iter().enumerate() { let Some(shard) = shard else { continue }; - let (rows, _) = load_rows(layout, &root, shard, id as u8).await?; + let (rows, _) = load_rows(layout, &root, shard, id as u8, None).await?; for binding in &rows { let pin = binding .control @@ -274,7 +277,7 @@ pub(in crate::node::bundle) async fn binding_inventory( session: SessionId, head: NodeBundleHead, ) -> Result> { - let Some(root) = load_root(layout, session, head).await? else { + let Some(root) = load_root(layout, session, head, None).await? else { return Ok( super::super::store::load_legacy_catalog(layout, session, head) .await? @@ -285,7 +288,7 @@ pub(in crate::node::bundle) async fn binding_inventory( let mut pins = std::collections::HashSet::new(); for (id, shard) in root.shards.iter().enumerate() { let Some(shard) = shard else { continue }; - let (rows, _) = load_rows(layout, &root, shard, id as u8).await?; + let (rows, _) = load_rows(layout, &root, shard, id as u8, None).await?; for binding in rows { let pin = binding .control diff --git a/crates/cellule-runtime/src/node/bundle/index/tests/admission.rs b/crates/cellule-runtime/src/node/bundle/index/tests/admission.rs index 210c0826..bccbf948 100644 --- a/crates/cellule-runtime/src/node/bundle/index/tests/admission.rs +++ b/crates/cellule-runtime/src/node/bundle/index/tests/admission.rs @@ -91,7 +91,7 @@ async fn aggregate_history_budget_is_checked_before_the_first_history_read() { .unwrap(); counted.reset(); assert!(matches!( - load_cells(&layout, session, head, &cells).await, + load_cells(&layout, session, head, &cells, None).await, Err(Error::Capacity("bundle selected history bytes")) )); assert_eq!( diff --git a/crates/cellule-runtime/src/node/bundle/mod.rs b/crates/cellule-runtime/src/node/bundle/mod.rs index 96332663..9c9e1e4f 100644 --- a/crates/cellule-runtime/src/node/bundle/mod.rs +++ b/crates/cellule-runtime/src/node/bundle/mod.rs @@ -41,12 +41,13 @@ mod binding; mod closure; mod codec; mod index; +mod origin; mod proof; pub(crate) mod recovery; mod selection; #[cfg(test)] use proof::checkpoint_prefix; -use proof::{verify_base, verify_binding}; +use proof::verify_base; pub(crate) use selection::confirm_selected_coverage; #[cfg(test)] use store::load_catalog; @@ -59,8 +60,8 @@ const MAX_BINDINGS: usize = 4_096; const MAX_LOCATORS: usize = 256; const MAX_INLINE_LOCATORS: usize = 32; const MAX_FRAMES: usize = 64; -// Verification retains at most one Cell suffix, independently of the number -// of historical objects referenced by its locators. +// Reconstruction retains at most one Cell suffix. Selection streams checked +// historical frames and shares one fresh cohort body for its new extents. const MAX_SUFFIX_BYTES: u64 = MAX_BUNDLE_BYTES; const MAX_BASE_OBJECTS: usize = 65_536; diff --git a/crates/cellule-runtime/src/node/bundle/origin.rs b/crates/cellule-runtime/src/node/bundle/origin.rs new file mode 100644 index 00000000..79b87a8a --- /dev/null +++ b/crates/cellule-runtime/src/node/bundle/origin.rs @@ -0,0 +1,80 @@ +//! One fresh, bounded object read shared by an individual selection operation. +use super::*; + +pub(super) struct OriginBundle { + session: SessionId, + head: NodeBundleHead, + body: Bytes, +} + +impl OriginBundle { + pub(super) async fn load( + layout: &cellule_ltx::CellStorageLayout, + prepared: &PreparedNodeBundle, + ) -> Result { + let (body, _) = layout + .store() + .get_with_etag_bounded( + &layout.node_coverage_bundle_path( + prepared.catalog.session.as_bytes(), + prepared.head.epoch, + prepared.head.digest.as_bytes(), + ), + MAX_BUNDLE_BYTES, + ) + .await?; + // Uploaded proposal bytes alone confer no availability. This fresh + // origin observation must match all bytes, not just the index header. + if body != prepared.body || index::body_digest(&body)? != prepared.head.digest { + return Err(Error::Node("bundle proposal differs from canonical origin")); + } + Ok(Self { + session: prepared.catalog.session, + head: prepared.head, + body, + }) + } + + fn range( + &self, + session: SessionId, + epoch: u64, + object: Digest, + range: &std::ops::Range, + ) -> Result> { + if session != self.session || epoch != self.head.epoch || object != self.head.digest { + return Ok(None); + } + if range.start > range.end || range.end > self.body.len() as u64 { + return Err(Error::Node("bundle extent is truncated")); + } + let start = + usize::try_from(range.start).map_err(|_| Error::Node("bundle extent overflow"))?; + let end = usize::try_from(range.end).map_err(|_| Error::Node("bundle extent overflow"))?; + Ok(Some(self.body.slice(start..end))) + } +} + +pub(super) async fn read_range( + layout: &cellule_ltx::CellStorageLayout, + session: SessionId, + epoch: u64, + object: Digest, + range: std::ops::Range, + origin: Option<&OriginBundle>, +) -> Result { + if let Some(body) = origin + .map(|origin| origin.range(session, epoch, object, &range)) + .transpose()? + .flatten() + { + return Ok(body); + } + Ok(layout + .store() + .range_get( + &layout.node_coverage_bundle_path(session.as_bytes(), epoch, object.as_bytes()), + range, + ) + .await?) +} diff --git a/crates/cellule-runtime/src/node/bundle/proof.rs b/crates/cellule-runtime/src/node/bundle/proof.rs index c91cf1ec..c19ca8ba 100644 --- a/crates/cellule-runtime/src/node/bundle/proof.rs +++ b/crates/cellule-runtime/src/node/bundle/proof.rs @@ -114,6 +114,42 @@ pub(super) async fn verify_binding( binding: &Binding, limits: cellule_ltx::Limits, ) -> Result> { + let mut frames = Vec::with_capacity(binding.locators.len()); + verify_binding_into( + layout, + session, + epoch, + binding, + limits, + None, + Some(&mut frames), + ) + .await?; + Ok(frames) +} + +pub(super) async fn verify_selected_binding( + layout: &cellule_ltx::CellStorageLayout, + session: SessionId, + epoch: u64, + binding: &Binding, + limits: cellule_ltx::Limits, + origin: &origin::OriginBundle, +) -> Result<()> { + // Selection needs exact locators, not retained native bodies. Keep the same + // verifier as reconstruction while dropping each checked frame promptly. + verify_binding_into(layout, session, epoch, binding, limits, Some(origin), None).await +} + +async fn verify_binding_into( + layout: &cellule_ltx::CellStorageLayout, + session: SessionId, + epoch: u64, + binding: &Binding, + limits: cellule_ltx::Limits, + origin: Option<&origin::OriginBundle>, + mut frames: Option<&mut Vec>, +) -> Result<()> { verify_base(layout, binding, limits).await?; let mut position = binding .control @@ -126,7 +162,6 @@ pub(super) async fn verify_binding( .as_ref() .ok_or(Error::Node("bundle base absent"))? .commit_sequence; - let mut frames = Vec::with_capacity(binding.locators.len()); let mut sequence = 0; let mut first_commit = commit; for locator in &binding.locators { @@ -137,13 +172,8 @@ pub(super) async fn verify_binding( .offset .checked_add(locator.bytes) .ok_or(Error::Node("bundle locator overflow"))?; - let bytes = layout - .store() - .range_get( - &layout.node_coverage_bundle_path(session.as_bytes(), epoch, object.as_bytes()), - locator.offset..end, - ) - .await?; + let bytes = + origin::read_range(layout, session, epoch, object, locator.offset..end, origin).await?; if bytes.len() as u64 != locator.bytes || *blake3::hash(&bytes).as_bytes() != *locator.frame_digest.as_bytes() { @@ -170,7 +200,9 @@ pub(super) async fn verify_binding( position = frame.segment().position(); commit = scope.commit_sequence; sequence = scope.node_sequence; - frames.push(frame); + if let Some(frames) = &mut frames { + frames.push(frame); + } } if position != binding.selected_position || commit != binding.selected_commit @@ -178,7 +210,7 @@ pub(super) async fn verify_binding( { return Err(Error::Node("bundle proof endpoint differs")); } - Ok(frames) + Ok(()) } pub(super) async fn verify_base( diff --git a/crates/cellule-runtime/src/node/bundle/selection.rs b/crates/cellule-runtime/src/node/bundle/selection.rs index c29420aa..04c80da5 100644 --- a/crates/cellule-runtime/src/node/bundle/selection.rs +++ b/crates/cellule-runtime/src/node/bundle/selection.rs @@ -232,11 +232,13 @@ impl NodeDirectory { ) }) .collect(); - let catalog = store::load_catalog_cells( + let origin = origin::OriginBundle::load(&self.layout, prepared).await?; + let catalog = index::load_cells( &self.layout, prepared.catalog.session, prepared.head, &cells, + Some(&origin), ) .await?; let mut proofs = Vec::new(); @@ -247,12 +249,13 @@ impl NodeDirectory { )) { continue; } - verify_binding( + proof::verify_selected_binding( &self.layout, catalog.session, catalog.epoch, &binding, limits, + &origin, ) .await?; let pin = binding diff --git a/crates/cellule-runtime/src/node/bundle/store.rs b/crates/cellule-runtime/src/node/bundle/store.rs index c0b9a279..2e9f1f97 100644 --- a/crates/cellule-runtime/src/node/bundle/store.rs +++ b/crates/cellule-runtime/src/node/bundle/store.rs @@ -166,7 +166,7 @@ pub(super) async fn load_catalog_cells( head: NodeBundleHead, cells: &std::collections::BTreeSet, ) -> Result { - index::load_cells(layout, session, head, cells).await + index::load_cells(layout, session, head, cells, None).await } pub(super) async fn load_legacy_catalog( diff --git a/crates/cellule-runtime/src/node/bundle/tests/index/cohort.rs b/crates/cellule-runtime/src/node/bundle/tests/index/cohort.rs index fe1f06aa..5f49d116 100644 --- a/crates/cellule-runtime/src/node/bundle/tests/index/cohort.rs +++ b/crates/cellule-runtime/src/node/bundle/tests/index/cohort.rs @@ -1,5 +1,78 @@ use super::*; +#[tokio::test] +async fn selection_reads_one_fresh_cohort_object_instead_of_each_new_extent() { + let mut f = Fixture::new().await; + super::super::coverage::enroll(&mut f).await; + let mut cells = Vec::new(); + for number in 4..(4 + MAX_FRAMES as u8) { + cells.push(f.cell(number).await); + f.heartbeat().await; + } + let mut frames = Vec::new(); + let mut assignments = Vec::new(); + for cell in &mut cells { + let (_, capture, assignment) = f.append(cell, 2); + frames.extend(capture); + assignments.push(assignment); + } + let now = f.node.advertisement().issued_at_ms(); + let prepared = f + .directory + .prepare_node_bundle(&f.node, &frames, &assignments, now) + .await + .unwrap(); + let path = f.layout.node_coverage_bundle_path( + f.node.advertisement().session().as_bytes(), + prepared.head.epoch, + prepared.head.digest.as_bytes(), + ); + f.count.reset(); + let (node, proofs) = f + .directory + .select_node_bundle(&f.node, &prepared, &f.lease, Limits::default(), now) + .await + .unwrap(); + let requests = f + .count + .requests() + .into_iter() + .filter(|request| request.location == path.as_ref()) + .collect::>(); + assert_eq!(proofs.len(), MAX_FRAMES); + assert_eq!(node.advertisement().bundle_head(), Some(prepared.head)); + eprintln!( + "fresh cohort object: cells={} reads={}", + cells.len(), + requests.len() + ); + assert_eq!( + requests.len(), + 1, + "read the complete fresh object once per selection" + ); + assert!(matches!( + requests[0].kind, + cellule_store::test_support::ObjectReadKind::Full + )); + for proof in &proofs { + assert_eq!(proof.commit_sequence(), 2); + assert_eq!(proof.locator_count(), 1); + } + f.node = node; + // The read is scoped to the operation. Neither an earlier live selection + // nor its prepared bytes can authorize a retry after origin disappears. + f.count.block_body_reads_for(&path); + let puts = f.count.put_requests(); + assert!( + f.directory + .select_node_bundle(&f.node, &prepared, &f.lease, Limits::default(), now) + .await + .is_err() + ); + assert_eq!(f.count.put_requests(), puts); +} + #[tokio::test] async fn checkpoint_cohort_uses_two_puts_and_preserves_a_hot_cell_suffix() { let mut f = Fixture::new().await; diff --git a/crates/cellule-runtime/src/node/durability/publication/mod.rs b/crates/cellule-runtime/src/node/durability/publication/mod.rs index af4ea129..3a397885 100644 --- a/crates/cellule-runtime/src/node/durability/publication/mod.rs +++ b/crates/cellule-runtime/src/node/durability/publication/mod.rs @@ -81,7 +81,9 @@ impl Publisher { .ok_or(Error::PendingPublication)?; resources.try_reserve( crate::fleet::resource::ResourceCost::zero() - .with_retained_bytes((4 * crate::node::bundle::MAX_BUNDLE_BYTES) as usize), + // The fifth bounded buffer is the fresh cohort origin read; + // its bytes are shared only within one verification operation. + .with_retained_bytes((5 * crate::node::bundle::MAX_BUNDLE_BYTES) as usize), ) } From ec127bad13d536ab6145e37d9b9056b71ca174c6 Mon Sep 17 00:00:00 2001 From: forhappy Date: Thu, 8 Oct 2026 15:33:08 -0700 Subject: [PATCH 042/102] Record fresh bundle read optimization and unqualified write measurements --- .../docs/write-performance-design.md | 9 + crates/cellule-runtime/src/node/bundle/mod.rs | 4 +- docs/bundle-coverage-implementation.md | 7 + docs/pr67-cohort-origin-measurement.md | 164 ++++++++++++++++++ docs/pr67-coverage-race-measurement.md | 6 +- docs/pr67-managed-producer-measurement.md | 2 + docs/write-performance-delivery.md | 13 +- 7 files changed, 201 insertions(+), 4 deletions(-) create mode 100644 docs/pr67-cohort-origin-measurement.md diff --git a/crates/cellule-runtime/docs/write-performance-design.md b/crates/cellule-runtime/docs/write-performance-design.md index ed8d110e..c1d91429 100644 --- a/crates/cellule-runtime/docs/write-performance-design.md +++ b/crates/cellule-runtime/docs/write-performance-design.md @@ -74,6 +74,15 @@ availability and drain. It is experimental, not performance qualification. The subsequent [coverage-race measurement](../../../docs/pr67-coverage-race-measurement.md) at `e40ecd6` passes warm/cold ACK read/retry and joined drain, but completes 100.20 Fleet writes/s versus 106.05 before the fix. There is no measured speedup. +The latest [cohort-origin comparison](../../../docs/pr67-cohort-origin-measurement.md) +at `9d4e632` reduces 187 reads of one fresh 64-Cell bundle to one. In one paired +window it completes 95.35 Fleet writes/s versus 88.28, with successful scheduled +p99 of 4,414.52 ms. ACK audits and drain pass, but total GET/range work remains +near 20.4 requests per completed write, root density is 1.08, and steady Bundle +ACKs remain zero. The positive paired rate difference does not establish a +repeatable gain; every point fails qualification. Its separately admitted +origin buffer raises the producer reservation from 16 to 20 MiB under the same +64-MiB diagnostic workload ledger. Per-Cell materializers, dense scheduling, complete failed-owner orchestration, large-capture fallback, retryable producer failures and collection remain open. Do not advance the follower reclamation frontier before diff --git a/crates/cellule-runtime/src/node/bundle/mod.rs b/crates/cellule-runtime/src/node/bundle/mod.rs index 9c9e1e4f..c31646a9 100644 --- a/crates/cellule-runtime/src/node/bundle/mod.rs +++ b/crates/cellule-runtime/src/node/bundle/mod.rs @@ -7,8 +7,8 @@ //! The caller owns host admission and the original node lease. This selection //! helper operates on complete captures assigned by the canonical shipper; //! live selection can confirm the original assigned captures locally without -//! another CAS. Ordinary actor bundle responses remain disabled until their -//! command/read/retry visibility and capture release consume that exact proof. +//! another CAS. Actor responses require the original matching proof; their +//! command/read/retry visibility and capture release consume that exact coverage. //! //! ```no_run //! use cellule_runtime::node::{NodeDirectory, VersionedNodeAdvertisement}; diff --git a/docs/bundle-coverage-implementation.md b/docs/bundle-coverage-implementation.md index 21f12ac0..b7590a6a 100644 --- a/docs/bundle-coverage-implementation.md +++ b/docs/bundle-coverage-implementation.md @@ -11,6 +11,13 @@ Fleet writes/s from 525.67 and failed availability/drain. The subsequent [coverage-race measurement](pr67-coverage-race-measurement.md) at `e40ecd6` passes all-ACK warm/cold audit and joined drain, but completes 100.20 Fleet writes/s versus 106.05 before the fix: no demonstrated throughput gain. +The latest [cohort-origin comparison](pr67-cohort-origin-measurement.md) at +`9d4e632` reduces one fresh 64-Cell bundle's origin reads from 187 to one. +It completes 95.35 Fleet writes/s versus 88.28 in a fresh paired window, with +passing ACK audits and drain. Total GET/range work remains near 20.4 requests +per completed write, root density is 1.08, and steady Bundle ACKs remain zero. +The producer charges its new buffer under the unchanged retention budget. +Every point fails qualification; a repeatable throughput gain remains unproved. This slice establishes ordering and reconstruction evidence. The [earlier application-path benchmark](pr67-performance-reevaluation.md) measures `7fc0793`; it does not exercise bundle-based responses or establish diff --git a/docs/pr67-cohort-origin-measurement.md b/docs/pr67-cohort-origin-measurement.md new file mode 100644 index 00000000..fe705ca5 --- /dev/null +++ b/docs/pr67-cohort-origin-measurement.md @@ -0,0 +1,164 @@ +# Fresh bundle origin reads: write parity still fails + +**Measured code: 95.35 Fleet writes/s and 274.53 Bucket writes/s.** In one +fresh paired diagnostic, Fleet completed 8.0% more writes and successful +scheduled p99 fell 46.1%. The Bucket fixture bypasses this optimization yet +completed 19.1% more writes; its request p99 worsened 2.9%. These short runs +establish neither a repeatable, attributable throughput gain nor parity. +**Every point fails qualification. PR #67 remains a draft.** + +## Verified change + +Candidate: `9d4e6328bceb2f88087698809c3b4167f94959ef`. +Immediate predecessor: `e40ecd6b6b2972ee91a52846737cbc12bd48f406`. +Celld: `f2bf648663a610eefde71f3547ad61e9b896b1f0`, using the unchanged pinned +image. Both Cellule release binaries were exported from committed source. + +Selection previously fetched a fresh cohort's header, shards, histories and +native frames separately. The real 64-Cell regression observed **187 reads of +the same newly uploaded object** before this change and **one full origin read** +after it. Selection now reads that object once, bounded to 4 MiB, compares every +byte with the proposal, checks its digest, and uses shared byte slices through +the same canonical index/frame verifiers. It drops checked frame bodies +promptly. Historical extents and each Cell's base dependencies still require +origin verification. The operation retains no availability cache across calls; +the regression also blocks the origin object and verifies selection retry fails. + +The fresh origin body's separate **4-MiB buffer is charged before installing the producer**. +The producer reservation increases **16 → 20 MiB** within the unchanged +64-MiB runtime retention ledger. The protocol object ceiling remains 4 MiB. +This admission cost reduces headroom for other retained work. No persisted +format, lease check, exact-range check or recovery requirement changes. + +## Workload and environment + +All six cases ran sequentially with fresh prefixes and fresh Linux provider +volumes. Comparison verified byte-identical client/auditor binaries, fixtures, +pinned images, loaded runner and Docker resource contract. Each used 1,000 +uniform Cells, 96-byte values, SQL INSERT plus SELECT with a two-hour result +ledger, 128 clients and 128 queue slots. Warmup was 30 seconds, followed by +one 60-second measured window. Fleet offered 15,000 writes/s with two followers; +Bucket offered 2,000/s without followers. These are SQL application diagnostics; +they do not reproduce the reported bounded-KV laptop workload. + +The ARM64 Docker VM has **8 CPUs and 8 GiB total shared RAM**. Serving containers +have an 8-CPU/16-GiB ceiling and 4-GiB tmpfs; the client ceiling is 4 CPUs/4 GiB. +RustFS has 2 CPUs and an 8-GiB memory/swap ceiling, using the same external +diagnostic adaptation for every arm. Container ceilings exceed available VM +resources; another workstation VM was present. The dedicated 8-CPU/16-GiB node, +2,000 Cells, three paired five-minute repetitions, physical-media durability, +and separate read/mixed qualification remain unverified. Acceptance gates and +workload budgets were preserved. + +## Reconciled write windows + +TPS counts successful logical writes completed inside the measured window. +Successful p99 is independent nearest-rank journal replay including trailing +successes. Scheduled latency starts at offered arrival; request latency starts +at issuance. Successful percentiles exclude errors and drops; those failures +remain explicit. The unchanged delivery gate also checks all-attempt latency. + +| Mode / system | Successful writes/s | Successful scheduled p99 ms | Successful request p99 ms | Errors | Queue drops | +| --- | ---: | ---: | ---: | ---: | ---: | +| Fleet / predecessor | 88.28 | 8,188.28 | 4,911.10 | 0 | 894,447 | +| Fleet / candidate | 95.35 | 4,414.52 | 2,800.05 | 0 | 894,023 | +| Fleet / celld | 4,076.40 | 252.99 | 71.66 | 48,999 | 606,417 | +| Bucket / predecessor | 230.58 | 5,783.21 | 3,713.34 | 0 | 105,909 | +| Bucket / candidate | 274.53 | 4,367.63 | 3,820.18 | 0 | 103,272 | +| Bucket / celld | 1,310.63 | 1,152.28 | 509.51 | 0 | 41,106 | + +Replay reconciled offers to successes, errors or drops, including trailing +completions, and reconciled every ACK cohort and its hash. **All six points fail +delivery qualification. These are overloaded completion rates, not sustainable +capacities.** The predecessor previously completed 100.20 Fleet/s and 271.63 +Bucket/s in the [coverage-race measurement](pr67-coverage-race-measurement.md). +Different runs of unchanged code vary substantially. The positive paired +differences above do not reverse the producer's earlier regression or establish +an attributable improvement. + +## Availability, drain and recovery + +| Mode / system | ACK cohort | Warm read/retry | Bucket-only cold read/retry | Drain seconds | +| --- | ---: | --- | --- | ---: | +| Fleet / predecessor | 7,371 | pass | pass | 27.98 | +| Fleet / candidate | 10,979 | pass | pass | 26.01 | +| Fleet / celld | 467,118 | fail: 463,546 HTTP errors | not reached | failed | +| Bucket / predecessor | 22,837 | pass | pass | 15.15 | +| Bucket / candidate | 28,125 | pass | pass | 21.26 | +| Bucket / celld | 118,680 | pass | pass | 1.75 | + +ACK cohorts include setup, warmup, steady and trailing successes. Passing audits +read and retry every original ACK and verify its stored result and incarnation; +these cases also pass the cold contract retry. Cold recovery starts with empty +local state after joined graceful drain. It does not qualify failed-owner +recovery before materialization or recovery of an owner-lost Fleet suffix. + +Celld Fleet's retained Docker state reports **OOMKilled: true, exit 137**. Its +warm audit failed and cold audit was not reached. That is an availability and +recovery-evidence failure in this shared-memory fixture, not established data +loss. Missing drain/cold evidence stays failed rather than being inferred. + +## Remaining bottleneck + +| Fleet window metric | Predecessor | Candidate | +| --- | ---: | ---: | +| Native-authority GET/range attempts per completed write | 6.32 | 3.04 | +| Total GET/range attempts per completed write | 20.47 | 20.40 | +| Successful PUTs per completed write | 3.30 | 3.68 | +| Materialized commands per selected root | 1.09 | 1.08 | +| Steady Bundle ACKs | 0 | 0 | +| Mean capture ms | 0.30 | 0.27 | +| Mean follower proof ms | 5.93 | 5.88 | +| Mean publication ms | 9,121.96 | 5,811.08 | +| Mean Fleet response ms | 1,850.17 | 1,282.76 | + +Native range reads fall **6.26 → 2.97 per completed write**, but immutable GETs +rise **12.24 → 15.44**. Operation counts include background work; phase samples +cover different cohorts and cannot be summed into one request's critical path. +Both windows still select almost one Cell root per command. The candidate +releases no additional commands through steady Bundle ACKs. Reducing one +verification object's fanout therefore does not remove ordinary per-Cell +publication or its upstream admission pressure. + +Candidate retained bytes sampled **22.98 → 62.34 MiB** against the unchanged +64-MiB ledger, including the 20-MiB producer reservation. Pending native object +sequences grow **435 → 470**; pending publications are **596 → 599**, with oldest +unpublished work **5,639 → 6,164 ms**. Two boundary samples do not establish +stable bounded debt. Activation took **70.80 seconds versus 45.06** for the +predecessor, excluded from write TPS but retained as a separate result. + +The next implementation must separate exact selected-capture cleanup from +immediate Cell-root materialization, retain admitted authenticated root debt, +and schedule fair dense checkpoints with joined shutdown and valid lease +renewal. The conditional **215 commands/checkpoint** and **0.05 publication +PUTs/command** targets remain unchanged. Bucket producer integration, +complete failed-owner suffix recovery including prior Fleet ACKs, safe +cross-Cell collection, and full read/write/mixed qualification remain open. + +## Verification and evidence + +The exact frozen functional candidate passed all eleven contributor routes on +Rust 1.97: format, targets/features, workspace tests (**1,956 passed, zero failed, +38 documented ignored**), local LTX, Clippy with `-D warnings`, API docs, +boundaries, layout, Rust fences, links and SQL/peer contracts. Linux releases +used the unchanged pinned Rust 1.98.1 image. The failing before-change I/O +regression is retained alongside passing after-change verification. + +Raw material stays outside Git under +`/Volumes/Workspace/crabbuild-target/cellule-write-perf-8ad1`. + +| Evidence | Identity | +| --- | --- | +| Frozen verification source manifest | `9a734c09d69a304204bc785bafc6b079e0c6c87cccc8aad87b34c4a03352e543` | +| Candidate release source manifest | `350915bcff5cbb06cac4e954a443be2fac81a39d8ae1877fac854c538693e5c6` | +| Candidate SQL binary | `1660e2f686a3f4bcb4fe22dea0068045bead27a2c9147999710ac92967faf386` | +| Identical client | `417f07b0424d27df75b1dca22666a7adb621db3cd11fdb1048eb9247a664e60b` | +| Identical auditor | `6595c24b0be217e181e8a9aa7de5cf478669ac341b37b76754fdcf0b3865b7b2` | +| Verified evidence index: 1,941 files | `60e1b32284fdce2cb7e67e22564ddf2f8a318f47e6b13cb288df660350d2393a` | + +The index is `cohort-origin-20261008-evidence-index.json`; its 1,203,520,400 bytes +were rehashed without mismatch. Replay, telemetry and paired comparisons are +`cohort-origin-20261008-{reconciled,telemetry,fleet-comparison,bucket-comparison}.json`. +Both comparison files verify fixture/host/runner provenance and explicitly +report `qualification_pass: false`. Linux provider volumes remain retained via +their recorded volume metadata. diff --git a/docs/pr67-coverage-race-measurement.md b/docs/pr67-coverage-race-measurement.md index c8cbf359..3a75306b 100644 --- a/docs/pr67-coverage-race-measurement.md +++ b/docs/pr67-coverage-race-measurement.md @@ -1,6 +1,10 @@ # Coverage race fix: measured recovery improvement, no throughput gain -**Latest measured code: 100.20 Fleet writes/s and 271.63 Bucket writes/s.** +Historical measurement. The subsequent +[cohort-origin comparison](pr67-cohort-origin-measurement.md) measures `9d4e632` +at 95.35 Fleet writes/s and 274.53 Bucket writes/s. Parity remains unqualified. + +**Code measured here: 100.20 Fleet writes/s and 271.63 Bucket writes/s.** Compared with the immediately preceding code, Fleet completed 5.5% fewer writes and Bucket 5.3% fewer in these single overloaded pairs. Successful scheduled p99 increased 9.3% and 14.2%, respectively. There is no demonstrated throughput or diff --git a/docs/pr67-managed-producer-measurement.md b/docs/pr67-managed-producer-measurement.md index 34d6c29d..42ecd398 100644 --- a/docs/pr67-managed-producer-measurement.md +++ b/docs/pr67-managed-producer-measurement.md @@ -3,6 +3,8 @@ Historical measurement. The subsequent [coverage-race fix and fresh comparison](pr67-coverage-race-measurement.md) restore Fleet warm/cold availability and drain, but demonstrate no TPS gain. +The latest [cohort-origin comparison](pr67-cohort-origin-measurement.md) +measures the next I/O reduction; write parity remains unqualified. **The new Fleet connection regressed. Parity is not achieved.** It completed 93.57 successful writes/s versus 525.67 before the producer: **82.2% lower in diff --git a/docs/write-performance-delivery.md b/docs/write-performance-delivery.md index 517a477e..769e7006 100644 --- a/docs/write-performance-delivery.md +++ b/docs/write-performance-delivery.md @@ -6,7 +6,18 @@ comparison retains exact retry and cold-state audits. **Celld write parity has not been established.** The [proposal](write-performance-proposal.md) remains the acceptance contract; completing tests or a load run does not pass its gates. -The latest [coverage-race measurement](pr67-coverage-race-measurement.md) +The latest [cohort-origin measurement](pr67-cohort-origin-measurement.md) +records runtime commit `9d4e632`: 95.35 Fleet writes/s and 274.53 Bucket writes/s +versus 88.28 and 230.58 for the immediate predecessor in fresh paired windows. +Fleet successful scheduled p99 is 4,414.52 ms; Bucket is 4,367.63 ms. The +64-Cell regression reduces reads of one fresh bundle from 187 to one; native +GET/range work falls, but total Fleet GET/range work remains near 20.4 requests +per completed write. Steady Bundle ACKs are zero and root density is 1.08. +All Cellule ACK audits and joined drains pass. Single short pairs, including a +Bucket fixture that bypasses the optimization, do not establish attributable +throughput gains. Every point fails qualification; PR #67 remains a draft. + +The earlier [coverage-race measurement](pr67-coverage-race-measurement.md) records runtime commit `e40ecd6`: 100.20 Fleet writes/s and 271.63 Bucket writes/s, 5.5% and 5.3% lower than the immediately preceding code in one fresh pair. Successful scheduled p99 also worsened. Fleet warm availability, joined drain From 6d62d4117712877f2c250d58f9b1cd60b529819b Mon Sep 17 00:00:00 2001 From: forhappy Date: Thu, 8 Oct 2026 16:19:13 -0700 Subject: [PATCH 043/102] Retire verified captures and materialize selected roots asynchronously --- .../src/cell/actor/admission.rs | 11 + .../cellule-runtime/src/cell/actor/group.rs | 3 +- .../cellule-runtime/src/cell/actor/handle.rs | 6 +- .../src/cell/actor/inventory/tests.rs | 2 + .../src/cell/actor/lifecycle/scheduling.rs | 3 +- .../src/cell/actor/materialization/mod.rs | 228 ++++++++++++++++++ crates/cellule-runtime/src/cell/actor/mod.rs | 5 +- .../src/cell/actor/requests.rs | 23 +- .../cellule-runtime/src/cell/actor/state.rs | 40 ++- crates/cellule-runtime/src/cell/actor/task.rs | 27 ++- .../src/cell/actor/tasks/activation.rs | 2 + .../src/cell/actor/tasks/mod.rs | 18 ++ .../src/cell/actor/tasks/publication.rs | 87 ++++++- .../src/cell/actor/tasks/work.rs | 2 +- .../src/cell/executor/bundle.rs | 37 ++- .../src/cell/executor/group.rs | 16 +- .../cellule-runtime/src/cell/executor/mod.rs | 11 +- crates/cellule-runtime/src/cell/worker/mod.rs | 4 +- crates/cellule-runtime/src/node/bundle/mod.rs | 46 ++++ .../src/node/bundle/tests/actor.rs | 14 +- .../src/node/bundle/tests/managed.rs | 163 +++++++++++-- crates/cellule-runtime/src/publication/mod.rs | 4 + 22 files changed, 661 insertions(+), 91 deletions(-) create mode 100644 crates/cellule-runtime/src/cell/actor/materialization/mod.rs diff --git a/crates/cellule-runtime/src/cell/actor/admission.rs b/crates/cellule-runtime/src/cell/actor/admission.rs index 9421c1cc..8510e3c6 100644 --- a/crates/cellule-runtime/src/cell/actor/admission.rs +++ b/crates/cellule-runtime/src/cell/actor/admission.rs @@ -64,6 +64,17 @@ pub(super) fn fence_active(active: &mut ActiveCell) { .refuse(DrainBlocker::IncompleteObservation, Error::Fenced); } active.inventory_refreshing = false; + // An accepted materializer owns its obligation until joined completion. + // Otherwise retain the original owner/pin via unpublished_node_logs while + // the fenced worker is discarded; no successful release is fabricated. + if !active.materializing && active.root_debt.take().is_some() { + active + .coordination + .step(CoordinationInput::FinishPublication { + fenced: true, + succeeded: false, + }); + } while let Some(publication) = active.publications.pop_front() { active .coordination diff --git a/crates/cellule-runtime/src/cell/actor/group.rs b/crates/cellule-runtime/src/cell/actor/group.rs index 275d1d6d..dcd764b6 100644 --- a/crates/cellule-runtime/src/cell/actor/group.rs +++ b/crates/cellule-runtime/src/cell/actor/group.rs @@ -220,7 +220,8 @@ pub(super) fn reply(command: &mut QueuedCommand, result: crate::Result u64 { self.expected_commit_sequence } - /// Returns the due time this Cell published. + /// Returns the due time covered by that exact selected commit. #[must_use] pub const fn next_due_ms(&self) -> i64 { self.next_due_ms diff --git a/crates/cellule-runtime/src/cell/actor/inventory/tests.rs b/crates/cellule-runtime/src/cell/actor/inventory/tests.rs index 3c0df482..d6082598 100644 --- a/crates/cellule-runtime/src/cell/actor/inventory/tests.rs +++ b/crates/cellule-runtime/src/cell/actor/inventory/tests.rs @@ -104,6 +104,8 @@ async fn stale_actor_probe_cannot_clear_newer_mutation_markers_or_replace_newer_ publisher: Some(publisher), publications: VecDeque::new(), publishing_since: None, + root_debt: None, + materializing: false, publication_bytes: 0, unpublished_node_logs: 0, queue: VecDeque::new(), diff --git a/crates/cellule-runtime/src/cell/actor/lifecycle/scheduling.rs b/crates/cellule-runtime/src/cell/actor/lifecycle/scheduling.rs index df1e18f5..4b8efc5f 100644 --- a/crates/cellule-runtime/src/cell/actor/lifecycle/scheduling.rs +++ b/crates/cellule-runtime/src/cell/actor/lifecycle/scheduling.rs @@ -8,7 +8,8 @@ pub(in crate::cell::actor) fn schedule( ) -> CoordinationDecision { let publication_blocked = active.queue.front().is_some_and(|work| { matches!(work, QueuedWork::Command(_)) - && (active.coordination.publication_count() >= MAX_PENDING_PUBLICATIONS + && (super::super::materialization::blocks_commands(active) + || active.coordination.publication_count() >= MAX_PENDING_PUBLICATIONS || active.publication_bytes >= PENDING_PUBLICATION_HIGH_WATER_BYTES) }); active.coordination.step(CoordinationInput::Schedule { diff --git a/crates/cellule-runtime/src/cell/actor/materialization/mod.rs b/crates/cellule-runtime/src/cell/actor/materialization/mod.rs new file mode 100644 index 00000000..224201fa --- /dev/null +++ b/crates/cellule-runtime/src/cell/actor/materialization/mod.rs @@ -0,0 +1,228 @@ +//! Exact shared selection and fair, admitted Cell-root materialization. + +use super::*; + +pub(super) const CHECKPOINT_COMMANDS: u64 = 215; +const MAX_ROOT_AGE: std::time::Duration = std::time::Duration::from_secs(45); +const MAX_MATERIALIZERS: usize = 8; + +pub(super) fn blocks_commands(active: &ActiveCell) -> bool { + active.root_debt.as_ref().is_some_and(|debt| { + debt.selected.proof.locator_count() >= CHECKPOINT_COMMANDS as usize + || debt + .selected + .proof + .native_suffix_bytes() + .is_ok_and(|bytes| bytes >= 3 << 20) + }) +} + +pub(super) fn start_selection( + cell: CellId, + active: &mut ActiveCell, + pool: &SqlWorkerPool, + tasks: &mut JoinSet, + publisher: CellPublisher, +) { + let coverage: Vec<_> = active.publications.drain(..).collect(); + let covered = coverage.len() as u64; + let retained_bytes = coverage + .iter() + .map(|queued| queued.pending.retained_bytes()) + .sum(); + let generation = active.generation; + let effect_id = active.begin_task(CoordinationEffect::Publication); + active.publishing_since = coverage.first().map(|queued| queued.submitted_at); + // Worker cleanup is dispatched through the same owned pool as SQL. The + // publisher token excludes root preparation throughout exact selection. + let pool = pool.clone(); + tasks.spawn(async move { + let result = async { + let newest = coverage.last().ok_or(Error::PendingPublication)?; + let mut captures = Vec::with_capacity(coverage.len()); + for queued in &coverage { + let durability = queued + .durability + .as_ref() + .ok_or(Error::PendingPublication)?; + let capture = durability + .selected_capture_prefix() + .await? + .ok_or(Error::Control("managed selection lacks capture"))?; + captures.push(capture); + } + let selected = Arc::clone(captures.last().ok_or(Error::PendingPublication)?.selected()); + if publisher.control().value().bundle_binding != Some(selected.proof.binding()) + || selected.proof.commit_sequence() != newest.pending.outcome().commit_sequence() + || selected.proof.position() != newest.pending.cuts().position + { + return Err(Error::Fenced); + } + let released = pool.release_bundle_captures(cell, captures).await?; + if released.len() != coverage.len() + || released + .iter() + .zip(&coverage) + .any(|(outcome, queued)| outcome != queued.pending.outcome()) + { + return Err(Error::Control("selected result differs from queued commit")); + } + Ok(Box::new(SelectedPublication { + covered, + debt: RootDebt { + selected, + durability: newest + .durability + .as_ref() + .ok_or(Error::PendingPublication)? + .clone(), + submitted_at: coverage + .first() + .ok_or(Error::PendingPublication)? + .submitted_at, + next_due_ms: newest.pending.next_due_ms(), + node_log_bytes: retained_bytes, + covered_node_logs: covered, + }, + })) + }; + let result = tokio::time::timeout(FLEET_PUBLICATION_GRACE, result) + .await + .map_err(|_| Error::Deadline) + .and_then(|result| result); + // The original durability receipt owns the Bundle/Fleet ACK; these + // senders represent only the ordinary root fallback, never a new proof. + if let Err(error) = &result { + tracing::warn!(cell = ?cell, error = ?error, "exact shared selection failed"); + let _ = pool.fence(cell).await; + } + drop(coverage); + TaskResult::BundleSelected { + cell, + generation, + effect_id, + publisher: Box::new(publisher), + covered, + retained_bytes, + result, + } + }); +} + +pub(super) fn dispatch( + pool: &SqlWorkerPool, + cells: &mut HashMap, + tasks: &mut JoinSet, +) { + let running = cells.values().filter(|active| active.materializing).count(); + let now = std::time::Instant::now(); + let mut ready: Vec<_> = cells + .iter() + .filter_map(|(cell, active)| { + let debt = active.root_debt.as_ref()?; + let forced = active.draining() + || active + .queue + .front() + .is_some_and(|work| matches!(work, QueuedWork::Migration(_))) + || !active.publications.is_empty() + && active.publications.iter().any(|queued| { + queued + .durability + .as_ref() + .is_none_or(|pending| !pending.has_managed_bundle_capture()) + }); + (active.publisher.is_some() + && !active.materializing + && (forced + || blocks_commands(active) + || debt + .selected + .proof + .commit_sequence() + .saturating_sub(active.published_sequence) + >= CHECKPOINT_COMMANDS + || now.saturating_duration_since(debt.submitted_at) >= MAX_ROOT_AGE)) + .then_some((debt.submitted_at, *cell)) + }) + .collect(); + ready.sort_unstable_by(|(a, cell_a), (b, cell_b)| { + a.cmp(b) + .then_with(|| cell_a.as_bytes().cmp(cell_b.as_bytes())) + }); + let mut available = MAX_MATERIALIZERS.saturating_sub(running); + for (_, cell) in ready { + if available == 0 { + break; + } + let Some(active) = cells.get_mut(&cell) else { + continue; + }; + let Some(debt) = active.root_debt.as_ref() else { + continue; + }; + let reservation = debt + .selected + .proof + .materialization_bytes() + .and_then(|bytes| { + pool.resource_ledger() + .try_reserve(ResourceCost::zero().with_retained_bytes(bytes)) + }); + let Ok(reservation) = reservation else { + continue; + }; + let Some(publisher) = active.publisher.take() else { + continue; + }; + let debt = debt.clone(); + active.materializing = true; + available -= 1; + let effect_id = active.begin_task(CoordinationEffect::Publication); + let generation = active.generation; + let published_sequence = active.published_sequence; + let pool = pool.clone(); + tasks.spawn(async move { + let _reservation = reservation; + let mut publisher = publisher; + let started = std::time::Instant::now(); + let commit_sequence = debt.selected.proof.commit_sequence(); + let result = async { + let deadline = started + FLEET_PUBLICATION_GRACE; + let mut delay = std::time::Duration::from_millis(100); + let root = loop { + match publisher.materialize_bundle_with_due(&debt.selected.proof, debt.next_due_ms).await { + Ok(root) => break root, + Err(error) if is_storage_publication_error(&error) && std::time::Instant::now() < deadline => { + tokio::time::sleep(delay).await; + delay = delay.saturating_mul(2).min(std::time::Duration::from_secs(2)); + } + Err(error) => return Err(error), + } + }; + // Checkpoint the authenticated locator prefix before admitting + // more selection; both root and index now cover this exact cut. + debt.durability.checkpoint_materialized(publisher.authority(), root).await?; + pool.bind_bundle_materialized(cell, root).await + }.await; + publisher.record_publication_timing(crate::fleet::telemetry::PublicationTiming { + queue_wait: started.saturating_duration_since(debt.submitted_at), + preparation: std::time::Duration::ZERO, + authority: started.elapsed(), total: debt.submitted_at.elapsed(), + succeeded: result.is_ok(), commit_sequence, + covered_commits: commit_sequence.saturating_sub(published_sequence), + }); + let fenced = result.is_err(); + if fenced { + tracing::warn!(cell = ?cell, error = ?result.as_ref().err(), "root materialization failed"); + let _ = pool.fence(cell).await; + } + TaskResult::Published { + cell, generation, effect_id, publisher: Box::new(publisher), + retained_bytes: 0, node_log_bytes: debt.node_log_bytes, + covered: 1, covered_node_logs: debt.covered_node_logs, + next_due_ms: debt.next_due_ms, commit_sequence, result, fenced, + } + }); + } +} diff --git a/crates/cellule-runtime/src/cell/actor/mod.rs b/crates/cellule-runtime/src/cell/actor/mod.rs index bfdcb402..70325f9d 100644 --- a/crates/cellule-runtime/src/cell/actor/mod.rs +++ b/crates/cellule-runtime/src/cell/actor/mod.rs @@ -32,6 +32,7 @@ mod bundle; mod group; mod lifecycle; mod maintenance; +mod materialization; pub use maintenance::MaintenanceCellRelease; mod receiver; mod requests; @@ -150,8 +151,8 @@ pub struct NodeJobReservation { /// command publications, excluding migration and failed recovery obligations. #[derive(Clone, Copy, Debug, PartialEq, Eq)] pub struct CellPublicationProgress { - /// Physical capture publications awaiting their terminal result. A capture - /// may contain several logically committed commands. + /// Physical captures and coalesced selected-root obligations awaiting their + /// terminal result. Either may contain several logical commands. pub pending_publications: usize, /// Retained capture bytes of those command publications. pub retained_capture_bytes: u64, diff --git a/crates/cellule-runtime/src/cell/actor/requests.rs b/crates/cellule-runtime/src/cell/actor/requests.rs index 24c56a6a..5a929e6b 100644 --- a/crates/cellule-runtime/src/cell/actor/requests.rs +++ b/crates/cellule-runtime/src/cell/actor/requests.rs @@ -392,6 +392,7 @@ pub(super) fn receive_publication_proof( pub(super) fn start_publication( cell: CellId, active: &mut ActiveCell, + pool: &SqlWorkerPool, tasks: &mut JoinSet, ) { let Some(mut publisher) = active.publisher.take() else { @@ -401,6 +402,19 @@ pub(super) fn start_publication( active.publisher = Some(publisher); return; } + if active.publications.iter().all(|queued| { + queued + .durability + .as_ref() + .is_some_and(PendingDurability::has_managed_bundle_capture) + }) { + super::materialization::start_selection(cell, active, pool, tasks, publisher); + return; + } + if active.root_debt.is_some() { + active.publisher = Some(publisher); + return; + } let generation = active.generation; let effect_id = active.begin_task(CoordinationEffect::Publication); let fleet_deadline = std::time::Instant::now() + FLEET_PUBLICATION_GRACE; @@ -557,7 +571,10 @@ pub(super) fn start_admitted_publication( // the sole publisher token and has not begun root preparation. // Drop the old admission before fresh origin reconstruction. drop(admitted.take()); - pool.release_bundle_captures(cell, captures).await?; + let released = pool.release_bundle_captures(cell, captures).await?; + if released != expected { + return Err(Error::Control("bundle result differs from queued commit")); + } drop(merged); for (mut pending, reservation) in pendings.drain(..).zip(retained_reservations.iter_mut()) @@ -593,11 +610,7 @@ pub(super) fn start_admitted_publication( if let Some(pending) = durabilities.last().and_then(Option::as_ref) { pending.checkpoint_materialized(publisher.authority(), root).await?; } - let published = pool.confirm_published_range(cell, root).await?; authority = authority_started.elapsed(); - if published != expected { - return Err(Error::Control("bundle result differs from queued commit")); - } return Ok(()); } let prepared = loop { diff --git a/crates/cellule-runtime/src/cell/actor/state.rs b/crates/cellule-runtime/src/cell/actor/state.rs index 4b7366b2..07c06269 100644 --- a/crates/cellule-runtime/src/cell/actor/state.rs +++ b/crates/cellule-runtime/src/cell/actor/state.rs @@ -100,7 +100,7 @@ pub(super) enum Message { require_resident: bool, reply: oneshot::Sender>, }, - /// Lists resident Cells whose published due time has passed. + /// Lists resident Cells whose selected durable due time has passed. /// /// The scheduler uses this to tick a Cell it already owns without reading /// its catalog entry or control record first. @@ -283,6 +283,8 @@ pub(super) struct ActiveCell { pub(super) publisher: Option, pub(super) durability_submitter: CellDurabilitySubmitter, pub(super) publications: VecDeque, + pub(super) root_debt: Option, + pub(super) materializing: bool, pub(super) publishing_since: Option, pub(super) publication_bytes: u64, pub(super) unpublished_node_logs: usize, @@ -383,7 +385,30 @@ pub(super) struct QueuedPublication { pub(super) proof: oneshot::Sender>, } +#[derive(Clone)] +pub(super) struct RootDebt { + pub(super) selected: Arc, + pub(super) durability: PendingDurability, + pub(super) submitted_at: std::time::Instant, + pub(super) next_due_ms: Option, + pub(super) node_log_bytes: u64, + pub(super) covered_node_logs: u64, +} + +pub(super) struct SelectedPublication { + pub(super) covered: u64, + pub(super) debt: RootDebt, +} + impl ActiveCell { + pub(super) fn selected_due_head(&self) -> (u64, Option) { + self.root_debt + .as_ref() + .map_or((self.published_sequence, self.next_due_ms), |debt| { + (debt.selected.proof.commit_sequence(), debt.next_due_ms) + }) + } + pub(super) fn draining(&self) -> bool { self.drain.is_some() || self.transfer.is_some() @@ -435,7 +460,7 @@ pub(super) struct LocalCell { pub(super) schema: u32, } -/// One resident Cell whose published due time has passed. +/// One resident Cell whose selected durable due time has passed. pub(super) struct DueResidentCell { pub(super) cell: CellId, pub(super) catalog: CatalogProof, @@ -443,7 +468,7 @@ pub(super) struct DueResidentCell { pub(super) code: Digest, pub(super) schema: u32, pub(super) admission: Arc, - /// Commit sequence the last authoritative publication named. + /// Commit sequence the last verified bundle or root named. pub(super) expected_commit_sequence: u64, pub(super) next_due_ms: i64, } @@ -517,6 +542,15 @@ pub(super) enum TaskResult { publisher: Box, result: crate::Result>, }, + BundleSelected { + cell: CellId, + generation: u64, + effect_id: u64, + publisher: Box, + covered: u64, + retained_bytes: u64, + result: crate::Result>, + }, Published { cell: CellId, generation: u64, diff --git a/crates/cellule-runtime/src/cell/actor/task.rs b/crates/cellule-runtime/src/cell/actor/task.rs index 2fbffc8e..bc8e222b 100644 --- a/crates/cellule-runtime/src/cell/actor/task.rs +++ b/crates/cellule-runtime/src/cell/actor/task.rs @@ -42,6 +42,7 @@ pub(super) async fn run( pressure_tick.set_missed_tick_behavior(tokio::time::MissedTickBehavior::Skip); pressure_tick.tick().await; loop { + super::materialization::dispatch(&pool, &mut cells, &mut tasks); if !shutdown.draining { maintenance::drive( &pool, @@ -379,12 +380,16 @@ pub(super) fn handle_message( oldest_unpublished: cells .values() .flat_map(|active| { - active.publishing_since.into_iter().chain( - active - .publications - .front() - .map(|queued| queued.submitted_at), - ) + active + .publishing_since + .into_iter() + .chain(active.root_debt.as_ref().map(|debt| debt.submitted_at)) + .chain( + active + .publications + .front() + .map(|queued| queued.submitted_at), + ) }) .min() .map(|oldest| now.saturating_duration_since(oldest)), @@ -630,14 +635,18 @@ pub(super) fn handle_message( .filter(|active| { active.transfer.is_none() && active.drain.is_none() - && active.next_due_ms.is_some_and(|due| due <= now_ms) + && active + .selected_due_head() + .1 + .is_some_and(|due| due <= now_ms) && matches!( active.coordination.lookup(), CoordinationDecision::LocalHandle ) }) .filter_map(|active| { - let next_due_ms = active.next_due_ms?; + let (expected_commit_sequence, next_due_ms) = active.selected_due_head(); + let next_due_ms = next_due_ms?; Some(DueResidentCell { cell: active.catalog.entry().cell(), catalog: active.catalog.clone(), @@ -645,7 +654,7 @@ pub(super) fn handle_message( code: active.code, schema: active.schema, admission: Arc::clone(&active.admission), - expected_commit_sequence: active.published_sequence, + expected_commit_sequence, next_due_ms, }) }) diff --git a/crates/cellule-runtime/src/cell/actor/tasks/activation.rs b/crates/cellule-runtime/src/cell/actor/tasks/activation.rs index 4fe65acb..e74ffd98 100644 --- a/crates/cellule-runtime/src/cell/actor/tasks/activation.rs +++ b/crates/cellule-runtime/src/cell/actor/tasks/activation.rs @@ -130,6 +130,8 @@ pub(super) fn handle_activated( durability_submitter, publications: VecDeque::new(), publishing_since: None, + root_debt: None, + materializing: false, publication_bytes: 0, unpublished_node_logs: 0, queue: VecDeque::new(), diff --git a/crates/cellule-runtime/src/cell/actor/tasks/mod.rs b/crates/cellule-runtime/src/cell/actor/tasks/mod.rs index b18fa509..672130f1 100644 --- a/crates/cellule-runtime/src/cell/actor/tasks/mod.rs +++ b/crates/cellule-runtime/src/cell/actor/tasks/mod.rs @@ -131,6 +131,24 @@ pub(super) fn handle_task( } => publication::handle_publication_admitted( context, cell, generation, effect_id, publisher, result, ), + TaskResult::BundleSelected { + cell, + generation, + effect_id, + publisher, + covered, + retained_bytes, + result, + } => publication::handle_bundle_selected( + context, + cell, + generation, + effect_id, + publisher, + covered, + retained_bytes, + result, + ), TaskResult::Published { cell, generation, diff --git a/crates/cellule-runtime/src/cell/actor/tasks/publication.rs b/crates/cellule-runtime/src/cell/actor/tasks/publication.rs index 6665c27f..db513e0b 100644 --- a/crates/cellule-runtime/src/cell/actor/tasks/publication.rs +++ b/crates/cellule-runtime/src/cell/actor/tasks/publication.rs @@ -2,6 +2,87 @@ use super::*; +pub(super) fn handle_bundle_selected( + context: TaskContext<'_>, + cell: CellId, + generation: u64, + effect_id: u64, + publisher: Box, + covered: u64, + retained_bytes: u64, + result: crate::Result>, +) { + let TaskContext { + pool, + cells, + transitioning, + tasks, + node_lease, + .. + } = context; + let Some(active) = cells.get_mut(&cell) else { + return; + }; + if active.generation != generation + || !active + .coordination + .effect_matches(effect_id, CoordinationEffect::Publication) + { + return; + } + active.finish_task(effect_id, CoordinationEffect::Publication); + active.publisher = Some(*publisher); + active.publishing_since = None; + active.publication_bytes = active.publication_bytes.saturating_sub(retained_bytes); + let result = result.and_then(|selected| { + node_lease.check()?; + if active.coordination.is_fenced() { + return Err(Error::Fenced); + } + Ok(selected) + }); + match result { + Ok(selected) => { + let mut debt = selected.debt; + let keep = if let Some(previous) = active.root_debt.take() { + debt.submitted_at = previous.submitted_at; + debt.node_log_bytes = debt.node_log_bytes.saturating_add(previous.node_log_bytes); + debt.covered_node_logs = debt + .covered_node_logs + .saturating_add(previous.covered_node_logs); + 0 + } else { + 1 + }; + active.root_debt = Some(debt); + // Keep one of the original publication obligations until root/index + // checkpoint completes; selection never makes the Cell releasable. + for _ in 0..selected.covered.saturating_sub(keep) { + active + .coordination + .step(CoordinationInput::FinishPublication { + fenced: false, + succeeded: true, + }); + } + start_publication(cell, active, pool, tasks); + } + Err(error) => { + tracing::warn!(cell = ?cell, error = ?error, "selected capture cleanup failed"); + for _ in 0..covered { + active + .coordination + .step(CoordinationInput::FinishPublication { + fenced: true, + succeeded: false, + }); + } + fence_active(active); + } + } + continue_cell(cell, pool, cells, transitioning, tasks, node_lease); +} + /// Applies a publication proof result and answers its waiter. pub(super) fn handle_proven( context: TaskContext<'_>, @@ -165,6 +246,10 @@ pub(super) fn handle_published( active.finish_task(effect_id, CoordinationEffect::Publication); active.last_work_at = std::time::Instant::now(); let object_published = result.is_ok(); + if active.materializing { + active.materializing = false; + active.root_debt = None; + } active.publishing_since = None; if object_published { // Control now names this commit, so the local mirror can answer a due @@ -204,7 +289,7 @@ pub(super) fn handle_published( if result.is_ok() { let _ = publications.send(active.catalog.entry().clone()); } - start_publication(cell, active, tasks); + start_publication(cell, active, pool, tasks); } continue_cell(cell, pool, cells, transitioning, tasks, node_lease); } diff --git a/crates/cellule-runtime/src/cell/actor/tasks/work.rs b/crates/cellule-runtime/src/cell/actor/tasks/work.rs index 994f49de..05e0a7f7 100644 --- a/crates/cellule-runtime/src/cell/actor/tasks/work.rs +++ b/crates/cellule-runtime/src/cell/actor/tasks/work.rs @@ -102,7 +102,7 @@ pub(super) fn handle_executed( active.unpublished_node_logs += 1; unpublished_node_log_bytes.fetch_add(retained_bytes, Ordering::AcqRel); } - start_publication(cell, active, tasks); + start_publication(cell, active, pool, tasks); let pool = pool.clone(); let generation = active.generation; let effect_id = active.begin_task(CoordinationEffect::Proof); diff --git a/crates/cellule-runtime/src/cell/executor/bundle.rs b/crates/cellule-runtime/src/cell/executor/bundle.rs index 7a1f30e6..d75d8698 100644 --- a/crates/cellule-runtime/src/cell/executor/bundle.rs +++ b/crates/cellule-runtime/src/cell/executor/bundle.rs @@ -7,9 +7,8 @@ impl CellExecutor { pub(crate) fn release_bundle_captures( &mut self, captures: &[VerifiedBundleCapture], - ) -> Result<()> { - if self.fenced || self.pending_migration.is_some() || self.bundle_materialization.is_some() - { + ) -> Result> { + if self.fenced || self.pending_migration.is_some() { return Err(Error::PendingPublication); } let newest = captures @@ -20,6 +19,9 @@ impl CellExecutor { .get(captures.len() - 1) .ok_or(Error::PendingPublication)?; let selected = newest.selected(); + if let Some(previous) = &self.bundle_materialization { + selected.proof.continues_selected_prefix(&previous.proof)?; + } if selected.proof.commit_sequence() != pending.outcome.commit_sequence() || selected.proof.position() != pending.cuts.position { @@ -41,19 +43,19 @@ impl CellExecutor { for pending in self.pending.iter().take(captures.len()) { self.db.prune_captured(&pending.cuts)?; } - // Keep outcomes and coordination pending until the actor's sole - // publisher materializes the exact endpoint. Proven visibility may - // advance now; an unproven later suffix still blocks queries/retries. - for pending in self.pending.iter_mut().take(captures.len()) { + // The original proof now owns this exact recoverable range. Outcomes + // remain in SQLite's durable retry ledger, rather than one heap entry + // per selected command. A later unproven suffix remains in `pending`. + let mut outcomes = Vec::with_capacity(captures.len()); + for pending in self.pending.drain(..captures.len()) { self.pending_bytes = self .pending_bytes .checked_sub(pending.retained_bytes()) .ok_or(Error::Control("bundle cleanup accounting underflow"))?; - pending.cuts.segments = Vec::new(); - pending.durable = true; + outcomes.push(pending.outcome); } self.bundle_materialization = Some(std::sync::Arc::clone(selected)); - Ok(()) + Ok(outcomes) } pub(crate) fn bind_bundle_materialized(&mut self, root: &cellule_ltx::RootRef) -> Result<()> { @@ -70,19 +72,8 @@ impl CellExecutor { "materialized root differs from bundle cleanup", )); } - let index = self - .pending - .iter() - .position(|pending| pending.outcome.commit_sequence() == root.commit_sequence) - .ok_or(Error::PendingPublication)?; - if self.pending.iter().take(index + 1).any(|pending| { - !pending.durable || !pending.cuts.segments.is_empty() || pending.prepared.is_some() - }) { - return Err(Error::Control("materialized bundle omits retained capture")); - } - for pending in self.pending.iter_mut().take(index + 1) { - pending.prepared = Some(*root); - } + self.published_sequence = root.commit_sequence; + self.bundle_materialization = None; Ok(()) } } diff --git a/crates/cellule-runtime/src/cell/executor/group.rs b/crates/cellule-runtime/src/cell/executor/group.rs index 343d99ad..2cf75b35 100644 --- a/crates/cellule-runtime/src/cell/executor/group.rs +++ b/crates/cellule-runtime/src/cell/executor/group.rs @@ -49,12 +49,16 @@ impl CellExecutor { if !self.accepts_publication() { return Err(Error::PendingPublication); } - let base_sequence = self - .pending - .back() - .map_or(self.published_sequence, |pending| { - pending.outcome.commit_sequence() - }); + let base_sequence = self.pending.back().map_or_else( + || { + self.bundle_materialization + .as_ref() + .map_or(self.published_sequence, |selected| { + selected.proof.commit_sequence() + }) + }, + |pending| pending.outcome.commit_sequence(), + ); let cell = self.cell; let incarnation = self.incarnation; let schema = self.schema; diff --git a/crates/cellule-runtime/src/cell/executor/mod.rs b/crates/cellule-runtime/src/cell/executor/mod.rs index c0ac57db..9ce2085d 100644 --- a/crates/cellule-runtime/src/cell/executor/mod.rs +++ b/crates/cellule-runtime/src/cell/executor/mod.rs @@ -865,7 +865,12 @@ impl CellExecutor { /// Marks one actor-ordered logical commit safe to observe before object publication. pub(crate) fn confirm_durable(&mut self, commit_sequence: u64) -> Result<()> { - if commit_sequence <= self.published_sequence { + if commit_sequence <= self.published_sequence + || self + .bundle_materialization + .as_ref() + .is_some_and(|selected| commit_sequence <= selected.proof.commit_sequence()) + { return Ok(()); } let index = self @@ -1281,7 +1286,9 @@ impl CellExecutor { } fn has_pending(&self) -> bool { - !self.pending.is_empty() || self.pending_migration.is_some() + !self.pending.is_empty() + || self.pending_migration.is_some() + || self.bundle_materialization.is_some() } fn accepts_publication(&self) -> bool { diff --git a/crates/cellule-runtime/src/cell/worker/mod.rs b/crates/cellule-runtime/src/cell/worker/mod.rs index ce4dd17d..7e8714dd 100644 --- a/crates/cellule-runtime/src/cell/worker/mod.rs +++ b/crates/cellule-runtime/src/cell/worker/mod.rs @@ -797,7 +797,7 @@ impl SqlWorkerPool { &self, cell: CellId, captures: Vec, - ) -> Result<()> { + ) -> Result> { let (reply, response) = oneshot::channel(); self.send( cell, @@ -1302,7 +1302,7 @@ enum WorkerCommand { ReleaseBundleCaptures { cell: CellId, captures: Vec, - reply: oneshot::Sender>, + reply: oneshot::Sender>>, }, BindBundleMaterialized { cell: CellId, diff --git a/crates/cellule-runtime/src/node/bundle/mod.rs b/crates/cellule-runtime/src/node/bundle/mod.rs index c31646a9..e0d6d308 100644 --- a/crates/cellule-runtime/src/node/bundle/mod.rs +++ b/crates/cellule-runtime/src/node/bundle/mod.rs @@ -160,6 +160,30 @@ pub struct BundleCoverageProof { live: Option, } impl BundleCoverageProof { + pub(crate) fn continues_selected_prefix(&self, previous: &Self) -> Result<()> { + if self.pin != previous.pin + || self.session != previous.session + || self.head.epoch != previous.head.epoch + || self.binding.application != previous.binding.application + || self.binding.control.cell != previous.binding.control.cell + || self.binding.control.incarnation != previous.binding.control.incarnation + || self.binding.control.epoch != previous.binding.control.epoch + || self.binding.control.code != previous.binding.control.code + || self.binding.control.schema != previous.binding.control.schema + || self.base()? != previous.base()? + || self.binding.selected_commit < previous.binding.selected_commit + || !self + .binding + .locators + .starts_with(&previous.binding.locators) + { + return Err(Error::Control( + "selected bundle does not continue root debt", + )); + } + Ok(()) + } + pub(crate) fn check_live_assignment( &self, assignment: &crate::node::log::AssignedCommitRange, @@ -186,6 +210,28 @@ impl BundleCoverageProof { pub fn locator_count(&self) -> usize { self.binding.locators.len() } + + pub(crate) fn materialization_bytes(&self) -> Result { + // Frame verification, overlay encoding and preparation may overlap. + // Reserve before the first origin read, including one bounded I/O body. + let native = self.native_suffix_bytes()?; + let bytes = native + .checked_mul(6) + .and_then(|bytes| bytes.checked_add(MAX_BUNDLE_BYTES)) + .ok_or(Error::Capacity("bundle materialization memory"))?; + usize::try_from(bytes).map_err(|_| Error::Capacity("bundle materialization memory")) + } + + pub(crate) fn native_suffix_bytes(&self) -> Result { + self.binding + .locators + .iter() + .try_fold(0_u64, |total, locator| { + total + .checked_add(locator.bytes) + .ok_or(Error::Capacity("bundle materialization memory")) + }) + } /// Exact immutable base required for reconstruction. pub fn base(&self) -> Result { self.binding diff --git a/crates/cellule-runtime/src/node/bundle/tests/actor.rs b/crates/cellule-runtime/src/node/bundle/tests/actor.rs index 789bbf10..b359dc07 100644 --- a/crates/cellule-runtime/src/node/bundle/tests/actor.rs +++ b/crates/cellule-runtime/src/node/bundle/tests/actor.rs @@ -446,15 +446,11 @@ async fn actor_case(prior_fleet: bool, cancel_caller: bool, early_selection: boo ) .await .unwrap(); - assert!( - pool.pending(target.cell_id()) - .await - .unwrap() - .unwrap() - .cuts() - .segments - .is_empty() - ); + assert!(pool.pending(target.cell_id()).await.unwrap().is_none()); + assert!(matches!( + pool.state(target.cell_id()).await.unwrap(), + crate::cell::worker::WorkerState::DurablePending + )); assert!( captured_paths.iter().all(|path| !path.exists()), "exact selected capture must release disk before root CAS" diff --git a/crates/cellule-runtime/src/node/bundle/tests/managed.rs b/crates/cellule-runtime/src/node/bundle/tests/managed.rs index 9d073f71..bdbd506c 100644 --- a/crates/cellule-runtime/src/node/bundle/tests/managed.rs +++ b/crates/cellule-runtime/src/node/bundle/tests/managed.rs @@ -167,6 +167,20 @@ async fn producer_failure_fences_new_work_and_join_preserves_its_cause() { #[tokio::test(flavor = "multi_thread")] async fn managed_producer_selects_actor_prefixes_and_joins_checkpoints_complete_close_and_cold_results() { + managed_actor_case(10, 32 << 20, false).await; +} + +#[tokio::test(flavor = "multi_thread")] +async fn managed_actor_retires_215_grouped_commands_per_cell_before_joined_root_materialization() { + managed_actor_case(215, 64 << 20, true).await; +} + +#[tokio::test(flavor = "multi_thread")] +async fn managed_actor_retires_65_sequential_captures_before_joined_root_materialization() { + managed_actor_case(65, 64 << 20, false).await; +} + +async fn managed_actor_case(per_cell: u16, retained_bytes: usize, grouped: bool) { let mut f = Fixture::new().await; super::coverage::enroll(&mut f).await; let authority = Arc::new(Authority { @@ -186,7 +200,7 @@ async fn managed_producer_selects_actor_prefixes_and_joins_checkpoints_complete_ let dirty = Arc::new(tokio::sync::Semaphore::new(1)); let runtime = CellRuntime::new_with_replica_host_requiring_node_lease( pool.clone(), - 32 << 20, + retained_bytes, SessionId::from_bytes([1; 16]), cellule_ltx::Host::default().with_dirty_slots(dirty.clone()), ) @@ -263,34 +277,77 @@ async fn managed_producer_selects_actor_prefixes_and_joins_checkpoints_complete_ // origin coverage to grant any command ACK, read or retry visibility. let preparation = dirty.acquire_owned().await.unwrap(); let digest = Digest::from_bytes([9; 32]); - for byte in 1_u8..=20 { - let handle = &cells[usize::from(byte % 2)].3; + let mut commands = tokio::task::JoinSet::new(); + let mut results = Vec::new(); + for byte in 1_u16..=per_cell * 2 { + if byte % 8 == 1 { + renew_actor_lease(&authority, &f.lease).await; + } + let handle = cells[usize::from(byte % 2)].3.clone(); + let mut request = [0; 16]; + request[..2].copy_from_slice(&byte.to_le_bytes()); let identity = MutationIdentity { - request_id: RequestId::from_bytes([byte; 16]), + request_id: RequestId::from_bytes(request), issued_at_ms: 10, expires_at_ms: 10_000, }; - let result = tokio::time::timeout( - std::time::Duration::from_secs(5), - handle.execute(identity, digest, 20, 1024, 1024, |tx| { - tx.execute("UPDATE counter SET value=value+1", [])?; - Ok(HandlerOutcome::Success(b"managed".to_vec())) - }), - ) - .await - .unwrap() - .unwrap(); - assert_eq!(result.commit_sequence(), u64::from(byte).div_ceil(2)); + let command = async move { + let result = tokio::time::timeout( + std::time::Duration::from_secs(5), + handle.execute(identity, digest, 20, 1024, 1024, |tx| { + tx.execute("UPDATE counter SET value=value+1", [])?; + Ok(HandlerOutcome::Success(b"managed".to_vec())) + }), + ) + .await + .unwrap() + .unwrap(); + (byte, identity, result) + }; + if grouped { + commands.spawn(command); + // Exercise native grouping within the unchanged 64-request Cell + // mailbox; this correctness test does not offer overload traffic. + if commands.len() == 32 { + while let Some(result) = commands.join_next().await { + results.push(result.unwrap()); + } + renew_actor_lease(&authority, &f.lease).await; + } + } else { + results.push(command.await); + } + } + while let Some(result) = commands.join_next().await { + results.push(result.unwrap()); + if results.len() % 8 == 0 { + renew_actor_lease(&authority, &f.lease).await; + } + } + for (byte, identity, result) in &results { + let handle = &cells[usize::from(byte % 2)].3; + if !grouped { + assert_eq!(result.commit_sequence(), u64::from(*byte).div_ceil(2)); + } assert_eq!( handle - .execute(identity, digest, 21, 1024, 1024, |_| panic!( + .execute(*identity, digest, 21, 1024, 1024, |_| panic!( "proved retry must not execute again" )) .await .unwrap(), - result + *result ); } + for parity in [0, 1] { + let mut sequences = results + .iter() + .filter(|(byte, _, _)| byte % 2 == parity) + .map(|(_, _, result)| result.commit_sequence()) + .collect::>(); + sequences.sort_unstable(); + assert_eq!(sequences, (1..=u64::from(per_cell)).collect::>()); + } for (_, _, _, handle) in &cells { assert_eq!( handle @@ -300,9 +357,48 @@ async fn managed_producer_selects_actor_prefixes_and_joins_checkpoints_complete_ .to_vec())) .await .unwrap(), - 10_i64.to_le_bytes() + i64::from(per_cell).to_le_bytes() ); } + // Root preparation is still held: exact selected reads/retries did not + // require any root, and covered physical captures have left worker RAM. + let mut selected_frames = 0_u64; + for (target, authority, _, _) in &cells { + assert_eq!( + authority + .load(target.cell_id()) + .await + .unwrap() + .unwrap() + .value() + .ltx_root() + .unwrap() + .commit_sequence, + 0 + ); + let proof = authority.load(target.cell_id()).await.unwrap().unwrap(); + let selected = f + .directory + .load_bundle_coverage(authority, &proof, Limits::default()) + .await + .unwrap(); + assert_eq!(selected.commit_sequence(), u64::from(per_cell)); + assert!(selected.locator_count() <= 256); + selected_frames += selected.locator_count() as u64; + tokio::time::timeout(std::time::Duration::from_secs(5), async { + while pool.pending(target.cell_id()).await.unwrap().is_some() { + tokio::task::yield_now().await; + } + }) + .await + .unwrap(); + } + let due = runtime.due_resident(i64::MAX, 2).await.unwrap(); + assert_eq!(due.len(), 2); + assert!( + due.iter() + .all(|cell| cell.expected_commit_sequence() == u64::from(per_cell)) + ); assert_eq!( responses .0 @@ -311,7 +407,7 @@ async fn managed_producer_selects_actor_prefixes_and_joins_checkpoints_complete_ .iter() .filter(|source| **source == CommandResponseSource::Bundle) .count(), - 20 + usize::from(per_cell) * 2 ); let shutdown = runtime.shutdown(); tokio::pin!(shutdown); @@ -336,7 +432,7 @@ async fn managed_producer_selects_actor_prefixes_and_joins_checkpoints_complete_ assert_eq!(control.value().state, ControlState::Idle); assert!(control.value().bundle_binding.is_none()); let root = control.value().ltx_root().unwrap(); - assert_eq!(root.commit_sequence, 10); + assert_eq!(root.commit_sequence, u64::from(per_cell)); let path = f .scratch .path() @@ -352,7 +448,7 @@ async fn managed_producer_selects_actor_prefixes_and_joins_checkpoints_complete_ assert_eq!( cold.query_row("SELECT value FROM counter", [], |row| row.get::<_, i64>(0)) .unwrap(), - 10 + i64::from(per_cell) ); assert_eq!( cold.query_row( @@ -361,10 +457,13 @@ async fn managed_producer_selects_actor_prefixes_and_joins_checkpoints_complete_ |row| row.get::<_, i64>(0) ) .unwrap(), - 10 + i64::from(per_cell) ); } - assert_eq!(durability.progress().unwrap().issued_through, 20); + assert_eq!( + durability.progress().unwrap().issued_through, + selected_frames + ); assert_eq!( pool.resource_ledger() .snapshot() @@ -374,3 +473,21 @@ async fn managed_producer_selects_actor_prefixes_and_joins_checkpoints_complete_ 0 ); } + +async fn renew_actor_lease(authority: &Authority, lease: &NodeLeaseGuard) { + // Advance the original guard only after a signed authoritative heartbeat. + let mut node = authority.observed.lock().await; + let mut next = node.advertisement().clone(); + next.issued_at_ms += 1; + next.expires_at_ms += 1; + next.progress += 1; + next.signature = SigningKey::from_bytes(&[10; 32]) + .sign(&next.signing_bytes().unwrap()) + .to_bytes(); + let now = next.issued_at_ms; + let refreshed = authority.directory.refresh(&node, next, now).await.unwrap(); + lease + .renew(now, refreshed.advertisement().expires_at_ms()) + .unwrap(); + *node = refreshed; +} diff --git a/crates/cellule-runtime/src/publication/mod.rs b/crates/cellule-runtime/src/publication/mod.rs index f2532209..4dcd311b 100644 --- a/crates/cellule-runtime/src/publication/mod.rs +++ b/crates/cellule-runtime/src/publication/mod.rs @@ -1241,6 +1241,10 @@ impl VerifiedBundleCapture { } impl PendingDurability { + pub(crate) fn has_managed_bundle_capture(&self) -> bool { + self.has_bundle_capture() && self.durability.managed_bundle_publication() + } + pub(crate) async fn selected_capture_prefix(&self) -> Result> { let Some(mut capture) = self.selected_capture().await? else { return Ok(None); From a0be6e9310c84e4864130c5c8d3526aeb4b461e9 Mon Sep 17 00:00:00 2001 From: forhappy Date: Thu, 8 Oct 2026 16:52:24 -0700 Subject: [PATCH 044/102] Record asynchronous-root measurements and availability regression --- .../docs/write-performance-design.md | 24 ++- docs/bundle-coverage-implementation.md | 41 ++-- docs/pr67-async-root-measurement.md | 186 ++++++++++++++++++ docs/pr67-cohort-origin-measurement.md | 4 + docs/write-performance-delivery.md | 16 +- 5 files changed, 249 insertions(+), 22 deletions(-) create mode 100644 docs/pr67-async-root-measurement.md diff --git a/crates/cellule-runtime/docs/write-performance-design.md b/crates/cellule-runtime/docs/write-performance-design.md index c1d91429..40a900f4 100644 --- a/crates/cellule-runtime/docs/write-performance-design.md +++ b/crates/cellule-runtime/docs/write-performance-design.md @@ -74,7 +74,7 @@ availability and drain. It is experimental, not performance qualification. The subsequent [coverage-race measurement](../../../docs/pr67-coverage-race-measurement.md) at `e40ecd6` passes warm/cold ACK read/retry and joined drain, but completes 100.20 Fleet writes/s versus 106.05 before the fix. There is no measured speedup. -The latest [cohort-origin comparison](../../../docs/pr67-cohort-origin-measurement.md) +The earlier [cohort-origin comparison](../../../docs/pr67-cohort-origin-measurement.md) at `9d4e632` reduces 187 reads of one fresh 64-Cell bundle to one. In one paired window it completes 95.35 Fleet writes/s versus 88.28, with successful scheduled p99 of 4,414.52 ms. ACK audits and drain pass, but total GET/range work remains @@ -83,8 +83,26 @@ ACKs remain zero. The positive paired rate difference does not establish a repeatable gain; every point fails qualification. Its separately admitted origin buffer raises the producer reservation from 16 to 20 MiB under the same 64-MiB diagnostic workload ledger. -Per-Cell materializers, dense scheduling, complete failed-owner orchestration, -large-capture fallback, retryable producer failures and collection remain open. +Managed selection now retires exact captures and coalesces one root obligation +per Cell. The worker keeps its latest authenticated selection and SQLite retry +results; sequence assignment includes the selected endpoint. Root jobs admit +memory before origin reads, order by oldest debt, permit at most eight jobs, and +join on shutdown. Logical 215-command density requests a checkpoint; physical +locator/byte pressure gates new commands. Root age of 45 seconds, drain, +migration and fallback also request materialization. Due hints expose the exact +selected head while a root lags. + +The latest [asynchronous-root comparison](../../../docs/pr67-async-root-measurement.md) +at `6d62d41` completes 183.65 Fleet writes/s versus 115.27, with successful +scheduled p99 of 1,782.74 ms. It returns 297,811 measured errors and fails its +warm ACK audit. No cold or successful drain evidence follows. Root density is +10.25 and PUT cost 0.48 per completed write; retained memory ends at 58.77 MiB +of 64 MiB and oldest debt reaches 49,054 ms. This is an availability regression, +not qualified improvement. The 215-command actor regression is grouped; only +65 sequential commands per Cell are covered by the passing held-root test. +Materializer progress under pressure, exact active-write checkpoint continuation, +application receipt visibility, complete failed-owner orchestration, large-capture +fallback, retryable producer failures and collection remain open. Do not advance the follower reclamation frontier before failed-owner recovery understands the selected bundle prefix. diff --git a/docs/bundle-coverage-implementation.md b/docs/bundle-coverage-implementation.md index b7590a6a..af25a6fa 100644 --- a/docs/bundle-coverage-implementation.md +++ b/docs/bundle-coverage-implementation.md @@ -11,7 +11,13 @@ Fleet writes/s from 525.67 and failed availability/drain. The subsequent [coverage-race measurement](pr67-coverage-race-measurement.md) at `e40ecd6` passes all-ACK warm/cold audit and joined drain, but completes 100.20 Fleet writes/s versus 106.05 before the fix: no demonstrated throughput gain. -The latest [cohort-origin comparison](pr67-cohort-origin-measurement.md) at +The latest [asynchronous-root comparison](pr67-async-root-measurement.md) at +`6d62d41` retires exact selected captures and schedules admitted roots separately. +It completes 183.65 Fleet writes/s versus 115.27 before, but returns 297,811 +measured request errors and fails its warm ACK audit. Cold recovery and successful +drain are unverified. Root density rises to 10.25, but retained pressure and +publication age grow. This is an availability regression; PR #67 is a draft. +The earlier [cohort-origin comparison](pr67-cohort-origin-measurement.md) at `9d4e632` reduces one fresh 64-Cell bundle's origin reads from 187 to one. It completes 95.35 Fleet writes/s versus 88.28 in a fresh paired window, with passing ACK audits and drain. Total GET/range work remains near 20.4 requests @@ -110,7 +116,7 @@ Its ordered gate prevents late assignment from consuming a node sequence. The legacy identity-free `DurabilityGate::issue` cannot produce a per-Cell closure: using it makes that closure fail closed. Managed close now waits for the complete issued producer prefix and joins exact checkpoint callbacks before Cell departure. -Failed-actor closure and dense materializer scheduling remain unqualified. +Failed-actor closure and sustained materializer progress remain unqualified. ```mermaid sequenceDiagram @@ -188,8 +194,10 @@ Exhaustion rejects preparation while retaining the last proof. These are safety ceilings, **not performance-qualified policies**. Old native bodies are still reverified. The 215-command checkpoint density is now represented and measured for a small-image fixture; total lifecycle cost is not qualified. Host memory, -native-job admission, materializer fairness and joined scheduling remain caller -obligations. See the [node performance design](../crates/cellule-runtime/docs/write-performance-design.md). +native-job admission and materializer progress still require qualification. +The actor now owns bounded admitted scheduling and joins accepted jobs; its +latest application measurement fails availability. See the +[node performance design](../crates/cellule-runtime/docs/write-performance-design.md). ## Verification and remaining delivery @@ -371,14 +379,14 @@ descriptor/body digest, and checks the live proof for each whole assignment. It validates the complete prefix before deleting any local file. A matching endpoint or a cold proof alone cannot release capture retention. -The worker keeps outcomes and its selected proof pending. The actor drops its -duplicate capture indexes and returns their memory reservation while preserving -outcome admission. The same actor-owned publisher reconstructs from authenticated -origin locators, preserves the new command's due time, and uses ordinary root -lineage and Cell CAS. Storage retries retain the existing publication grace; -shutdown still joins materialization and complete original issued-range closure. -The installed feed gets a bounded 100-ms selection opportunity before the -ordinary root fallback. Root preparation already in progress retains its files. +For the managed producer, the worker removes selected physical captures and +heap outcome entries, retains one latest admitted proof, and keeps retry results +in SQLite. The actor coalesces one root-debt obligation per Cell. Its original +publisher reconstructs admitted roots from authenticated locators, preserves due +time and uses ordinary lineage, Cell CAS and exact catalog checkpoint. Storage +retries retain the existing grace; shutdown joins accepted materializers and +complete original issued-range closure. The manual-feed/unleased fallback retains +its bounded 100-ms selection opportunity and ownership of files already preparing. The real actor test blocks root preparation until selection, then pauses root CAS and verifies that every selected capture file is absent. It issues a later @@ -390,10 +398,11 @@ reconstruction tests, not TPS qualification. Remaining work before qualified production enablement: -1. Build on the installed bounded producer: fix measured verification overhead - and failed-actor closure, then schedule fair admitted materializer cohorts, - coalescing proven debt rather than creating one root per command. Connect - Bucket-only publication and preserve exact capture/visibility gates. +1. Fix the measured Fleet availability regression. Prove admitted materializer + progress, exact checkpoint continuation under active writes and application + minimum-receipt read/retry visibility. Address verification overhead and + failed-actor closure. Connect Bucket-only publication while preserving exact + capture/visibility gates. 2. Build on the authenticated index: bound admitted maintenance inventory, increase checkpoint density with retained-byte accounting, and measure the complete materialization/checkpoint/collection cost. diff --git a/docs/pr67-async-root-measurement.md b/docs/pr67-async-root-measurement.md new file mode 100644 index 00000000..155b2193 --- /dev/null +++ b/docs/pr67-async-root-measurement.md @@ -0,0 +1,186 @@ +# Asynchronous roots: measured Fleet availability regression + +**Measured candidate: 183.65 Fleet writes/s and 204.10 Bucket writes/s.** +The Fleet candidate returns **297,811 measured request errors**, versus zero +before the change, and fails its warm ACK audit. Its higher completion rate and +lower successful-write p99 do not establish an acceptable improvement. +**Every point fails qualification. PR #67 is a draft and is not ready to merge.** + +## Tested implementation + +Candidate: `6d62d4117712877f2c250d58f9b1cd60b529819b`. +Functional baseline: `9d4e6328bceb2f88087698809c3b4167f94959ef`; its subsequent +`ec127bad` commit changes documentation only. Celld: +`f2bf648663a610eefde71f3547ad61e9b896b1f0`, using the unchanged pinned image. +Both Cellule releases were built from committed source. + +The managed actor now verifies and retires the complete oldest selected capture +prefix through its original publisher and native assignments. It keeps one +authenticated root-debt obligation per Cell, while the worker retains the latest +selected proof and keeps retry results in SQLite. SQL sequence assignment +includes that selected endpoint after physical captures leave the worker queue. +Grouped responses accept the same original Bundle proof as individual commands. + +Root materialization runs asynchronously through the canonical root and catalog +checkpoint paths. It admits memory before origin reads, orders candidates by +oldest debt and Cell identity, permits at most eight jobs, and joins accepted +jobs during shutdown. The logical checkpoint target is 215 commands; physical +locator/byte pressure also gates new commands. Root age of 45 seconds, drain, +migration and ordinary publication fallback request materialization. Due hints +use the exact selected head while a root lags. No persisted format or signed +message changes. + +Two-Cell tests hold root preparation and verify 65 sequential commands per Cell +or 215 grouped commands per Cell, reads, recorded retries, exact capture cleanup, +joined shutdown and cold SQLite contents. **The 215-command test is grouped**; +it does not establish 215 sequential physical captures under sustained load. +A longer sequential held-root test failed during development and remains in +external evidence. Active-write checkpoint continuation and application receipt +visibility under pressure still need regressions and successful integration runs. + +## Matched Docker workload + +All six cases ran sequentially with fresh prefixes and fresh Linux provider +volumes. Comparison verifies identical client/auditor binaries, fixtures, pinned +images, runner and Docker resource contract. Each uses 1,000 uniform Cells, +96-byte values, INSERT plus SELECT in the command, a two-hour durable result +ledger, 128 clients and 128 queue slots. Warmup is 30 seconds; each measured +window is 60 seconds with one repetition. Fleet offers 15,000 writes/s with two +followers. Bucket offers 2,000/s without followers. + +The ARM64 VM has **8 CPUs and 8 GiB total shared memory**. Serving containers +have an 8-CPU/16-GiB ceiling and 4-GiB tmpfs, the client 4 CPUs/4 GiB, and RustFS +2 CPUs/8 GiB using the unchanged external provider adaptation. Ceilings exceed +VM resources. Cellule keeps the same 64-MiB retention ledger and 1-GiB managed +disk budget. This is an overloaded SQL application diagnostic, not the bounded +KV laptop workload or dedicated 8-vCPU/16-GiB node qualification. Acceptance +gates were preserved; a short diagnostic cannot pass the five-minute gate. + +## Reconciled results + +TPS counts successful logical writes completed inside the measured window. +Successful p99 uses independent nearest-rank journal replay, including trailing +successes. Scheduled latency starts at offered arrival; request latency starts +at issuance. These successful percentiles exclude errors and drops, which stay +explicit. The qualification reporter also checks all-attempt latency. + +| Mode / system | Successful writes/s | Successful scheduled p99 ms | Successful request p99 ms | Measured errors | Queue drops | +| --- | ---: | ---: | ---: | ---: | ---: | +| Fleet / baseline | 115.27 | 4,538.19 | 2,560.72 | 0 | 892,828 | +| Fleet / candidate | 183.65 | 1,782.74 | 1,159.62 | 297,811 | 591,170 | +| Fleet / celld | 2,878.73 | 544.60 | 297.91 | 0 | 727,020 | +| Bucket / baseline | 196.27 | 6,566.99 | 4,797.93 | 0 | 107,968 | +| Bucket / candidate | 204.10 | 5,492.60 | 4,666.86 | 0 | 107,498 | +| Bucket / celld | 1,319.80 | 779.30 | 437.71 | 0 | 40,560 | + +The candidate also returns **27 warmup errors**; the other five cases return +zero warmup errors. All cases drop warmup offers. Replay reconciles every +generated measured offer to an attempt or drop, every attempt to success or +error, window/trailing completions, and each complete ACK cohort and digest. +**These are completion rates under overload, not sustainable capacities.** + +Fleet completes 59.3% more writes in this pair, but the errors and failed ACK +availability make the change a regression. Its successful latency statistics +describe a different surviving cohort and cannot establish overall improvement. +The Bucket fixture bypasses the managed producer, so its 4.0% paired difference +cannot be attributed to this implementation. The unchanged baseline previously +completed 95.35 Fleet/s and 274.53 Bucket/s in the +[cohort-origin measurement](pr67-cohort-origin-measurement.md); observed variance +remains outside the required A/A agreement. No repeatable gain or parity is proven. + +## Availability, drain and recovery + +| Mode / system | Complete ACK cohort | Warm read/retry | Cold read/retry | Successful drain seconds | +| --- | ---: | --- | --- | ---: | +| Fleet / baseline | 12,289 | pass | pass | 22.26 | +| Fleet / candidate | 20,689 | fail: 16,340 errors | not reached | no successful record | +| Fleet / celld | 335,996 | fail: 335,996 errors | not reached | no successful record | +| Bucket / baseline | 21,491 | pass | pass | 14.06 | +| Bucket / candidate | 22,766 | pass | pass | 10.58 | +| Bucket / celld | 114,418 | pass | pass | 0.88 | + +Passing cases read and retry every seed, contract, warmup, measured and trailing +ACK, and pass the cold contract retry. Cold restore starts with empty local +state after joined graceful drain; it does not qualify owner loss before +materialization or recovery of the complete owner-lost Fleet suffix. + +Candidate Fleet errors include HTTP 503 temporary unavailability. Its warm +audit checks all 20,689 records, successfully retries 4,349, and reports 16,340 +errors. The runner aborts before cold recovery; absent successful joined-drain +evidence remains failed. This does not establish acknowledged-data loss, but +acknowledged-data recovery is unverified for this candidate. + +Celld Fleet's owner exits with **code 3, OOMKilled false**. Its retained log +records ambiguous lease renewals followed by the node lease watchdog self-fence. +All warm queries fail and cold recovery is not reached. This differs from the +earlier OOM run; neither failure is omitted or converted into a passing result. +Provider startup byte/inode gates pass. The aborted Fleet cases lack later +filesystem/cold lifecycle observations, so their complete provider evidence also +fails. Missing observations do not prove the provider itself exited. + +## Publication work and remaining bottleneck + +| Fleet window metric | Baseline | Candidate | +| --- | ---: | ---: | +| Successful provider PUTs per completed write | 3.669 | 0.482 | +| Total GET/range attempts per completed write | 21.239 | 13.304 | +| Materialized commands per selected root | 1.120 | 10.247 | +| Steady Bundle ACKs | 0 | 0 | +| Mean capture ms | 0.239 | 0.190 | +| Mean follower-proof ms | 5.487 | 5.811 | +| End retained MiB / 64-MiB ledger | 22.41 | 58.77 | +| End oldest publication debt ms | 3,633 | 49,054 | +| End active Cells | 1,000 | 995 | + +The candidate reduces root/publication work in this window, but still misses +the conditional **215 commands/checkpoint** and **0.05 PUTs/command** targets. +Native-authority range attempts grow from 20,729 to 104,200 in absolute terms; +total GET/range work falls per completed write because immutable reads fall and +the denominator changes. Steady responses still use Fleet proofs; some Bundle +responses occur outside this window. Phase samples describe different cohorts +and cannot be summed into a request's critical path. + +Candidate retained bytes grow **27.93 → 58.77 MiB**, while accounted unpublished +native-log bytes grow **10.58 → 22.70 MiB**. Oldest root/publication debt grows +**43,786 → 49,054 ms**. Selected physical capture bytes fall, but total retained +pressure remains. New publication counts include coalesced root obligations, +so their counts are not directly comparable with the former per-capture counts. +Two boundary samples do not prove stable bounded debt or isolate the cause of +HTTP 503 failures. + +The next fixes must prove materializer admission/progress under the same ledger, +exact checkpoint continuation while writes remain active, and ordinary +application minimum-receipt read/retry availability while roots lag. Preserve +the failed snapshot and rerun matched measurements after each demonstrated +correction. Bucket producer integration, complete failed-owner issued-suffix +recovery including prior Fleet ACKs, safe cross-Cell collection, physical-device +durability and full read/write/mixed qualification remain open. + +## Verification and retained evidence + +The exact frozen functional source passes all eleven contributor routes on +Rust 1.97: format, features/targets, workspace tests (**1,958 passed, zero failed, +38 documented ignored**), local LTX, Clippy with warnings denied, API docs, +boundaries, layout, document fences/links and SQL/peer contracts. Linux releases +use the unchanged pinned Rust 1.98.1 image. These checks do not qualify performance. + +Raw material stays outside Git under +`/Volumes/Workspace/crabbuild-target/cellule-write-perf-8ad1`. + +| Evidence | SHA-256 | +| --- | --- | +| Frozen verification source manifest | `17ae0d679452b1ec00a7ba3589af22502928514c593e87f7b5dc533c0f251ee5` | +| Candidate release source manifest | `8879b6c26aad10432eb0781bbbdb0642e2de6662f50f6355cfd09e7666a85644` | +| Candidate SQL binary | `077503622744d5bd3f6cb5fc5de45bb74806ba4e35ef2a72546d140bda01b176` | +| Identical client | `417f07b0424d27df75b1dca22666a7adb621db3cd11fdb1048eb9247a664e60b` | +| Identical auditor | `6595c24b0be217e181e8a9aa7de5cf478669ac341b37b76754fdcf0b3865b7b2` | +| Evidence index: 5,762 files | `006d03a60f9505685a77e8de4863c89386b859f226f3c09945009af85af0f8fb` | + +`async-root-20261008-evidence-index.json` covers 1,219,891,080 bytes, including +source, immutable builds, all six cases, failed development logs and final +verification. Every indexed file was rehashed without mismatch. Independent +replay, telemetry and paired comparisons are +`async-root-20261008-{reconciled,telemetry,fleet-comparison,bucket-comparison}.json`. +Both comparisons verify fixture/host/runner provenance and report +`qualification_pass: false`. Provider volumes remain retained through recorded +metadata. No measurement process or serving container remains running. diff --git a/docs/pr67-cohort-origin-measurement.md b/docs/pr67-cohort-origin-measurement.md index fe705ca5..c5dc4313 100644 --- a/docs/pr67-cohort-origin-measurement.md +++ b/docs/pr67-cohort-origin-measurement.md @@ -1,5 +1,9 @@ # Fresh bundle origin reads: write parity still fails +This measures the earlier `9d4e632` implementation. The latest +[asynchronous-root measurement](pr67-async-root-measurement.md) records `6d62d41` +and its Fleet availability regression. + **Measured code: 95.35 Fleet writes/s and 274.53 Bucket writes/s.** In one fresh paired diagnostic, Fleet completed 8.0% more writes and successful scheduled p99 fell 46.1%. The Bucket fixture bypasses this optimization yet diff --git a/docs/write-performance-delivery.md b/docs/write-performance-delivery.md index 769e7006..d5aee747 100644 --- a/docs/write-performance-delivery.md +++ b/docs/write-performance-delivery.md @@ -6,7 +6,17 @@ comparison retains exact retry and cold-state audits. **Celld write parity has not been established.** The [proposal](write-performance-proposal.md) remains the acceptance contract; completing tests or a load run does not pass its gates. -The latest [cohort-origin measurement](pr67-cohort-origin-measurement.md) +The latest [asynchronous-root measurement](pr67-async-root-measurement.md) +records runtime commit `6d62d41`: 183.65 Fleet writes/s and 204.10 Bucket writes/s +versus 115.27 and 196.27 in fresh paired windows. Fleet successful scheduled p99 +is 1,782.74 ms, but the candidate returns 297,811 measured errors and fails its +warm ACK audit; cold recovery and successful drain are unverified. This is an +availability regression, not an acceptable performance gain. Root density rises +from 1.12 to 10.25 and PUTs fall from 3.67 to 0.48 per completed write, while +retention and publication age grow. Bucket audits pass, but its fixture bypasses +the producer. Every point fails qualification; PR #67 is a draft. + +The earlier [cohort-origin measurement](pr67-cohort-origin-measurement.md) records runtime commit `9d4e632`: 95.35 Fleet writes/s and 274.53 Bucket writes/s versus 88.28 and 230.58 for the immediate predecessor in fresh paired windows. Fleet successful scheduled p99 is 4,414.52 ms; Bucket is 4,367.63 ms. The @@ -103,7 +113,7 @@ collection paths. There is no legacy decoding or automatic migration. | M1 | Packs, inline leaves and bounded compaction spooling delivered | Ordinary append meets four PUTs; composed compaction needs five. Paired cold/sparse-read guardrail unverified | | M2 | Bounded file-backed shared publication coordinator implemented; exact scope, restore, cancellation, minimum-budget and dormant-sibling retention checks added | Earlier three active-Fleet windows cost 5.229–5.433 PUTs/command; two fail the debt trend. Per-Cell authority work remains; M4 is required | | M3 | Signed 512-sequence/five-second grants, bounded local registry, lifecycle gate and signed HTTP fixture implemented | Full isolated checks and native lifecycle suite pass; earlier three active-Fleet windows cost 0.0138 enrollment GETs/command. The latest 15K diagnostic still fails delivery despite passing ACK audits | -| M4 | [Connected protocol APIs](bundle-coverage-implementation.md), actor command/read/retry and exact capture release, retained bounded producer, fair native/checkpoint turns, monotonic coverage joining and joined closure | Latest Fleet warm/cold ACK audit and drain pass; throughput/latency targets still fail. Dense materializer scheduling, Bucket connection, failed-node orchestration, collection and qualification remain open | +| M4 | [Connected protocol APIs](bundle-coverage-implementation.md), exact capture retirement, coalesced root debt, admitted asynchronous materializers, retained producer, fair native/checkpoint turns and joined closure | Latest Fleet availability and ACK audit regress. Materializer progress, checkpoint continuation, application receipt visibility, Bucket connection, failed-node orchestration, collection and qualification remain open | | M5 | Three paired low-rate Fleet repetitions and target diagnostics with exact ACK audits delivered | Publication stability and target delivery fail; qualified capacity, read/failure/overload matrix and absolute/relative parity remain unverified | ## Shared publication checkpoint @@ -120,7 +130,7 @@ source identities, checks and performance results will be recorded separately. Signed append grants are implemented with a fresh issuance path and local durable closure gates. Bundle ACKs now require the installed original producer and admitted exact proof. Its Fleet connection remains experimental after the -measured regression; dense scheduling, full recovery orchestration, collection +measured regression; sustained materializer progress, full recovery orchestration, collection and qualification still need production integration. The first isolated M2/M3 snapshot passed 1,857 workspace tests (38 documented tests From 4a8f55cc99414e6a3907531c774a184842e1de3e Mon Sep 17 00:00:00 2001 From: forhappy Date: Thu, 8 Oct 2026 17:28:18 -0700 Subject: [PATCH 045/102] Preserve selected receipts across confirmed asynchronous checkpoints --- .../docs/write-performance-design.md | 10 + .../src/cell/actor/materialization/mod.rs | 14 +- .../src/cell/actor/requests.rs | 5 +- .../src/cell/executor/bundle.rs | 20 +- .../cellule-runtime/src/cell/executor/mod.rs | 5 + crates/cellule-runtime/src/cell/worker/mod.rs | 9 +- crates/cellule-runtime/src/cell/worker/run.rs | 9 +- crates/cellule-runtime/src/fleet/resource.rs | 34 +++ .../src/node/bundle/continuation.rs | 210 ++++++++++++++++++ crates/cellule-runtime/src/node/bundle/mod.rs | 27 +-- .../src/node/bundle/tests/index/checkpoint.rs | 137 +++++++++++- .../src/node/bundle/tests/managed.rs | 128 +++++++++-- docs/bundle-coverage-implementation.md | 6 + 13 files changed, 567 insertions(+), 47 deletions(-) create mode 100644 crates/cellule-runtime/src/node/bundle/continuation.rs diff --git a/crates/cellule-runtime/docs/write-performance-design.md b/crates/cellule-runtime/docs/write-performance-design.md index 40a900f4..3da1b9eb 100644 --- a/crates/cellule-runtime/docs/write-performance-design.md +++ b/crates/cellule-runtime/docs/write-performance-design.md @@ -92,6 +92,16 @@ locator/byte pressure gates new commands. Root age of 45 seconds, drain, migration and fallback also request materialization. Due hints expose the exact selected head while a root lags. +A follow-up preserves exact selected suffixes across a confirmed checkpoint, +including receipts selected against intermediate bases while older roots were +preparing. Process-local hash-chain witnesses retain no frame bodies and expire +when their original prefix exceeds the 256-locator proof bound. Their memory +transfers from admission acquired before root I/O and releases with the worker. +Regression cases exercise live writes, read/retry visibility, joined cold +restore and successive intermediate bases. This correctness change still needs +an end-to-end measurement; it supplies no new durability or origin-availability +proof. + The latest [asynchronous-root comparison](../../../docs/pr67-async-root-measurement.md) at `6d62d41` completes 183.65 Fleet writes/s versus 115.27, with successful scheduled p99 of 1,782.74 ms. It returns 297,811 measured errors and fails its diff --git a/crates/cellule-runtime/src/cell/actor/materialization/mod.rs b/crates/cellule-runtime/src/cell/actor/materialization/mod.rs index 224201fa..37d847ab 100644 --- a/crates/cellule-runtime/src/cell/actor/materialization/mod.rs +++ b/crates/cellule-runtime/src/cell/actor/materialization/mod.rs @@ -165,6 +165,13 @@ pub(super) fn dispatch( .selected .proof .materialization_bytes() + .and_then(|bytes| { + bytes + .checked_add( + crate::node::bundle::MaterializedBundlePrefix::maximum_retained_bytes(), + ) + .ok_or(Error::Capacity("checkpoint metadata admission")) + }) .and_then(|bytes| { pool.resource_ledger() .try_reserve(ResourceCost::zero().with_retained_bytes(bytes)) @@ -183,7 +190,7 @@ pub(super) fn dispatch( let published_sequence = active.published_sequence; let pool = pool.clone(); tasks.spawn(async move { - let _reservation = reservation; + let mut reservation = reservation; let mut publisher = publisher; let started = std::time::Instant::now(); let commit_sequence = debt.selected.proof.commit_sequence(); @@ -203,7 +210,10 @@ pub(super) fn dispatch( // Checkpoint the authenticated locator prefix before admitting // more selection; both root and index now cover this exact cut. debt.durability.checkpoint_materialized(publisher.authority(), root).await?; - pool.bind_bundle_materialized(cell, root).await + // Transfer pre-admitted metadata to the worker. No admission + // can fail after the canonical root and checkpoint are joined. + let retained = reservation.split_retained(crate::node::bundle::MaterializedBundlePrefix::maximum_retained_bytes())?; + pool.bind_bundle_materialized(cell, root, retained).await }.await; publisher.record_publication_timing(crate::fleet::telemetry::PublicationTiming { queue_wait: started.saturating_duration_since(debt.submitted_at), diff --git a/crates/cellule-runtime/src/cell/actor/requests.rs b/crates/cellule-runtime/src/cell/actor/requests.rs index 5a929e6b..2f06bcc1 100644 --- a/crates/cellule-runtime/src/cell/actor/requests.rs +++ b/crates/cellule-runtime/src/cell/actor/requests.rs @@ -587,6 +587,9 @@ pub(super) fn start_admitted_publication( } preparation = preparation_started.elapsed(); let authority_started = std::time::Instant::now(); + let checkpoint_retained = pool.resource_ledger().try_reserve( + ResourceCost::zero().with_retained_bytes(crate::node::bundle::MaterializedBundlePrefix::maximum_retained_bytes()) + )?; let root = loop { match publisher .materialize_bundle_with_due(&selected.proof, published_next_due_ms) @@ -605,11 +608,11 @@ pub(super) fn start_admitted_publication( Err(error) => return Err(error), } }; - pool.bind_bundle_materialized(cell, root).await?; PendingDurability::prove_objects(&durabilities).await?; if let Some(pending) = durabilities.last().and_then(Option::as_ref) { pending.checkpoint_materialized(publisher.authority(), root).await?; } + pool.bind_bundle_materialized(cell, root, checkpoint_retained).await?; authority = authority_started.elapsed(); return Ok(()); } diff --git a/crates/cellule-runtime/src/cell/executor/bundle.rs b/crates/cellule-runtime/src/cell/executor/bundle.rs index d75d8698..2399b514 100644 --- a/crates/cellule-runtime/src/cell/executor/bundle.rs +++ b/crates/cellule-runtime/src/cell/executor/bundle.rs @@ -20,7 +20,10 @@ impl CellExecutor { .ok_or(Error::PendingPublication)?; let selected = newest.selected(); if let Some(previous) = &self.bundle_materialization { - selected.proof.continues_selected_prefix(&previous.proof)?; + selected.proof.continues_selected_prefix( + &previous.proof, + self.bundle_checkpoint.as_ref().map(|(prefix, _)| prefix), + )?; } if selected.proof.commit_sequence() != pending.outcome.commit_sequence() || selected.proof.position() != pending.cuts.position @@ -58,7 +61,11 @@ impl CellExecutor { Ok(outcomes) } - pub(crate) fn bind_bundle_materialized(&mut self, root: &cellule_ltx::RootRef) -> Result<()> { + pub(crate) fn bind_bundle_materialized( + &mut self, + root: &cellule_ltx::RootRef, + mut retained: crate::fleet::resource::ResourceReservation, + ) -> Result<()> { let selected = self .bundle_materialization .as_ref() @@ -72,7 +79,16 @@ impl CellExecutor { "materialized root differs from bundle cleanup", )); } + // The actor has joined the exact root CAS and catalog checkpoint before + // this native bind. Retain a compact identity of that original prefix, + // so older receipts can meet later proofs rebased on this exact root. + let checkpoint = selected.proof.materialized_prefix( + *root, + self.bundle_checkpoint.as_ref().map(|(prefix, _)| prefix), + )?; + retained.shrink_retained(checkpoint.retained_bytes())?; self.published_sequence = root.commit_sequence; + self.bundle_checkpoint = Some((checkpoint, retained)); self.bundle_materialization = None; Ok(()) } diff --git a/crates/cellule-runtime/src/cell/executor/mod.rs b/crates/cellule-runtime/src/cell/executor/mod.rs index 9ce2085d..3740da51 100644 --- a/crates/cellule-runtime/src/cell/executor/mod.rs +++ b/crates/cellule-runtime/src/cell/executor/mod.rs @@ -298,6 +298,10 @@ pub struct CellExecutor { pending_bytes: u64, published_sequence: u64, bundle_materialization: Option>, + bundle_checkpoint: Option<( + crate::node::bundle::MaterializedBundlePrefix, + crate::fleet::resource::ResourceReservation, + )>, pending_migration: Option, fenced: bool, } @@ -351,6 +355,7 @@ impl CellExecutor { pending_bytes: 0, published_sequence: 0, bundle_materialization: None, + bundle_checkpoint: None, pending_migration: None, fenced: false, } diff --git a/crates/cellule-runtime/src/cell/worker/mod.rs b/crates/cellule-runtime/src/cell/worker/mod.rs index 7e8714dd..938b0852 100644 --- a/crates/cellule-runtime/src/cell/worker/mod.rs +++ b/crates/cellule-runtime/src/cell/worker/mod.rs @@ -815,11 +815,17 @@ impl SqlWorkerPool { &self, cell: CellId, root: cellule_ltx::RootRef, + retained: ResourceReservation, ) -> Result<()> { let (reply, response) = oneshot::channel(); self.send( cell, - WorkerCommand::BindBundleMaterialized { cell, root, reply }, + WorkerCommand::BindBundleMaterialized { + cell, + root, + retained, + reply, + }, ) .await?; receive(response).await @@ -1307,6 +1313,7 @@ enum WorkerCommand { BindBundleMaterialized { cell: CellId, root: cellule_ltx::RootRef, + retained: ResourceReservation, reply: oneshot::Sender>, }, ConfirmBootstrapPublished { diff --git a/crates/cellule-runtime/src/cell/worker/run.rs b/crates/cellule-runtime/src/cell/worker/run.rs index 8d786cfb..41b9b777 100644 --- a/crates/cellule-runtime/src/cell/worker/run.rs +++ b/crates/cellule-runtime/src/cell/worker/run.rs @@ -412,11 +412,16 @@ fn run_worker_command( .and_then(|cell| cell.executor.release_bundle_captures(&captures)); let _ = reply.send(result); } - WorkerCommand::BindBundleMaterialized { cell, root, reply } => { + WorkerCommand::BindBundleMaterialized { + cell, + root, + retained, + reply, + } => { let result = cells .get_mut(&cell) .ok_or(Error::CellNotActive) - .and_then(|cell| cell.executor.bind_bundle_materialized(&root)); + .and_then(|cell| cell.executor.bind_bundle_materialized(&root, retained)); let _ = reply.send(result); } WorkerCommand::ConfirmBootstrapPublished { cell, cuts, reply } => { diff --git a/crates/cellule-runtime/src/fleet/resource.rs b/crates/cellule-runtime/src/fleet/resource.rs index dc993cc3..eaf56a5d 100644 --- a/crates/cellule-runtime/src/fleet/resource.rs +++ b/crates/cellule-runtime/src/fleet/resource.rs @@ -491,6 +491,21 @@ pub(crate) struct ResourceReservation { } impl ResourceReservation { + /// Transfers already admitted memory to an independently owned lifetime. + /// Total ledger usage is unchanged; this cannot admit after publication. + pub(crate) fn split_retained(&mut self, bytes: usize) -> Result { + let remaining = self + .cost + .retained_bytes + .checked_sub(bytes) + .ok_or(Error::Capacity("retained transfer exceeds admission"))?; + self.cost.retained_bytes = remaining; + Ok(Self { + ledger: self.ledger.clone(), + cost: ResourceCost::zero().with_retained_bytes(bytes), + }) + } + /// Returns only memory whose owned capture indexes have already been dropped. pub(crate) fn shrink_retained(&mut self, bytes: usize) -> Result<()> { let released = self @@ -627,6 +642,25 @@ impl cellule_ltx::HostResourceAdmission for LedgerHostResourceAdmission { mod tests { use super::*; + #[test] + fn retained_transfer_preserves_usage_until_each_owner_releases() { + let cost = ResourceCost::active_cell().with_retained_bytes(128); + let ledger = ResourceLedger::new(cost); + let mut source = ledger.try_reserve(cost).unwrap(); + let mut checkpoint = source.split_retained(96).unwrap(); + assert_eq!(ledger.snapshot().unwrap().used, cost); + assert!(source.split_retained(33).is_err()); + checkpoint.shrink_retained(16).unwrap(); + assert_eq!(ledger.snapshot().unwrap().used.retained_bytes(), 48); + drop(source); + assert_eq!( + ledger.snapshot().unwrap().used, + ResourceCost::zero().with_retained_bytes(16) + ); + drop(checkpoint); + assert_eq!(ledger.snapshot().unwrap().used, ResourceCost::zero()); + } + #[test] fn capture_cleanup_returns_only_released_memory_and_preserves_outcome_admission() { let cost = ResourceCost::active_cell().with_retained_bytes(128); diff --git a/crates/cellule-runtime/src/node/bundle/continuation.rs b/crates/cellule-runtime/src/node/bundle/continuation.rs new file mode 100644 index 00000000..f7507b90 --- /dev/null +++ b/crates/cellule-runtime/src/node/bundle/continuation.rs @@ -0,0 +1,210 @@ +//! Exact selected-debt continuation across confirmed root checkpoints. +use super::*; + +struct PrefixAlias { + base: cellule_ltx::RootRef, + locators: usize, + digest: Digest, +} + +/// Process-local witnesses for the prefixes a confirmed root materialized. +/// These grant no coverage or origin availability and retain no frame bodies. +pub(crate) struct MaterializedBundlePrefix { + root: cellule_ltx::RootRef, + pin: BundleBindingRef, + aliases: Vec, +} + +impl MaterializedBundlePrefix { + pub(crate) const fn maximum_retained_bytes() -> usize { + MAX_LOCATORS * std::mem::size_of::() + } + + pub(crate) fn retained_bytes(&self) -> usize { + self.aliases.capacity() * std::mem::size_of::() + } +} + +fn extend_digest(mut digest: Digest, locators: &[Locator]) -> Result { + for locator in locators { + let mut hash = blake3::Hasher::new(); + hash.update(digest.as_bytes()); + hash.update( + locator + .object + .ok_or(Error::Control("checkpoint locator is unresolved"))? + .as_bytes(), + ); + hash.update(&locator.offset.to_le_bytes()); + hash.update(&locator.bytes.to_le_bytes()); + hash.update(locator.frame_digest.as_bytes()); + digest = Digest::from_bytes(*hash.finalize().as_bytes()); + } + Ok(digest) +} + +fn locator_prefix_digest(locators: &[Locator]) -> Result { + extend_digest( + Digest::from_bytes(*blake3::hash(b"cellule.confirmed-bundle-prefix.v1").as_bytes()), + locators, + ) +} + +impl BundleCoverageProof { + pub(crate) fn materialized_prefix( + &self, + root: cellule_ltx::RootRef, + previous: Option<&MaterializedBundlePrefix>, + ) -> Result { + let base = self.base()?; + if root.cell != *self.binding.control.cell.as_bytes() + || root.incarnation != *self.binding.control.incarnation.as_bytes() + || root.commit_sequence != self.commit_sequence() + || root.position != self.position() + || root.commit_sequence <= base.commit_sequence + || self.binding.locators.is_empty() + || self.binding.locators.len() > MAX_LOCATORS + { + return Err(Error::Control( + "materialized root differs from selected prefix", + )); + } + let Some(previous) = previous else { + return Ok(MaterializedBundlePrefix { + root, + pin: self.pin, + aliases: vec![PrefixAlias { + base, + locators: self.binding.locators.len(), + digest: locator_prefix_digest(&self.binding.locators)?, + }], + }); + }; + if previous.pin != self.pin + || root.commit_sequence <= previous.root.commit_sequence + || root.position.txid <= previous.root.position.txid + { + return Err(Error::Control( + "checkpoint does not advance original writer", + )); + } + let delta = if base == previous.root { + self.binding.locators.as_slice() + } else { + let alias = previous + .aliases + .iter() + .find(|alias| alias.base == base) + .ok_or(Error::Control("checkpoint lacks original base witness"))?; + let prefix = self + .binding + .locators + .get(..alias.locators) + .ok_or(Error::Control("checkpoint omits confirmed prefix"))?; + if locator_prefix_digest(prefix)? != alias.digest { + return Err(Error::Control("checkpoint changes confirmed prefix")); + } + &self.binding.locators[alias.locators..] + }; + if delta.is_empty() { + return Err(Error::Control("checkpoint lacks native advancement")); + } + // Every advancing checkpoint adds a native locator. An older base + // expires once no bounded proof can still contain its entire prefix. + // Extend hashes, rather than retain O(history) locator arrays. + let mut aliases = Vec::with_capacity((previous.aliases.len() + 1).min(MAX_LOCATORS)); + for alias in &previous.aliases { + let count = alias + .locators + .checked_add(delta.len()) + .ok_or(Error::Control("checkpoint prefix count overflow"))?; + if count <= MAX_LOCATORS { + aliases.push(PrefixAlias { + base: alias.base, + locators: count, + digest: extend_digest(alias.digest, delta)?, + }); + } + } + if aliases.len() >= MAX_LOCATORS { + return Err(Error::Control("checkpoint alias bound exceeded")); + } + aliases.push(PrefixAlias { + base: previous.root, + locators: delta.len(), + digest: locator_prefix_digest(delta)?, + }); + Ok(MaterializedBundlePrefix { + root, + pin: self.pin, + aliases, + }) + } + + pub(crate) fn continues_selected_prefix( + &self, + previous: &Self, + checkpoint: Option<&MaterializedBundlePrefix>, + ) -> Result<()> { + if self.pin != previous.pin + || self.session != previous.session + || self.head.epoch != previous.head.epoch + || self.binding.application != previous.binding.application + || self.binding.control.cell != previous.binding.control.cell + || self.binding.control.incarnation != previous.binding.control.incarnation + || self.binding.control.epoch != previous.binding.control.epoch + || self.binding.control.code != previous.binding.control.code + || self.binding.control.schema != previous.binding.control.schema + || self.binding.selected_commit < previous.binding.selected_commit + { + return Err(Error::Control( + "selected bundle does not continue root debt", + )); + } + let base = self.base()?; + let previous_base = previous.base()?; + let prefix = if base == previous_base { + previous.binding.locators.as_slice() + } else { + let checkpoint = checkpoint.ok_or(Error::Control( + "selected bundle changes an unconfirmed base", + ))?; + if checkpoint.root != base + || checkpoint.pin != previous.pin + || checkpoint.root.commit_sequence > previous.commit_sequence() + || checkpoint.root.position.txid > previous.position().txid + { + return Err(Error::Control("selected bundle changes the checkpoint")); + } + let alias = checkpoint + .aliases + .iter() + .find(|alias| alias.base == previous_base) + .ok_or(Error::Control( + "selected bundle lacks original base witness", + ))?; + let materialized = + previous + .binding + .locators + .get(..alias.locators) + .ok_or(Error::Control( + "selected bundle omits materialized locators", + ))?; + if locator_prefix_digest(materialized)? != alias.digest { + return Err(Error::Control( + "selected bundle changes materialized locators", + )); + } + // Only the original confirmed prefix may disappear. Compare the + // complete remaining suffix, not an endpoint watermark. + &previous.binding.locators[alias.locators..] + }; + if !self.binding.locators.starts_with(prefix) { + return Err(Error::Control( + "selected bundle does not continue root debt", + )); + } + Ok(()) + } +} diff --git a/crates/cellule-runtime/src/node/bundle/mod.rs b/crates/cellule-runtime/src/node/bundle/mod.rs index e0d6d308..978c7ec6 100644 --- a/crates/cellule-runtime/src/node/bundle/mod.rs +++ b/crates/cellule-runtime/src/node/bundle/mod.rs @@ -40,11 +40,13 @@ use bytes::Bytes; mod binding; mod closure; mod codec; +mod continuation; mod index; mod origin; mod proof; pub(crate) mod recovery; mod selection; +pub(crate) use continuation::MaterializedBundlePrefix; #[cfg(test)] use proof::checkpoint_prefix; use proof::verify_base; @@ -159,31 +161,8 @@ pub struct BundleCoverageProof { // Cold reconstruction must never revive the original process's ACK gate. live: Option, } -impl BundleCoverageProof { - pub(crate) fn continues_selected_prefix(&self, previous: &Self) -> Result<()> { - if self.pin != previous.pin - || self.session != previous.session - || self.head.epoch != previous.head.epoch - || self.binding.application != previous.binding.application - || self.binding.control.cell != previous.binding.control.cell - || self.binding.control.incarnation != previous.binding.control.incarnation - || self.binding.control.epoch != previous.binding.control.epoch - || self.binding.control.code != previous.binding.control.code - || self.binding.control.schema != previous.binding.control.schema - || self.base()? != previous.base()? - || self.binding.selected_commit < previous.binding.selected_commit - || !self - .binding - .locators - .starts_with(&previous.binding.locators) - { - return Err(Error::Control( - "selected bundle does not continue root debt", - )); - } - Ok(()) - } +impl BundleCoverageProof { pub(crate) fn check_live_assignment( &self, assignment: &crate::node::log::AssignedCommitRange, diff --git a/crates/cellule-runtime/src/node/bundle/tests/index/checkpoint.rs b/crates/cellule-runtime/src/node/bundle/tests/index/checkpoint.rs index 6653885e..0dcb2055 100644 --- a/crates/cellule-runtime/src/node/bundle/tests/index/checkpoint.rs +++ b/crates/cellule-runtime/src/node/bundle/tests/index/checkpoint.rs @@ -1,5 +1,98 @@ use super::*; +#[tokio::test] +async fn checkpoint_continuation_preserves_receipts_from_an_intermediate_base() { + let mut f = Fixture::new().await; + let mut cell = f.cell(4).await; + macro_rules! select { + ($sequence:literal) => {{ + let (_, frames, assigned) = f.append(&mut cell, $sequence); + let proposal = f + .directory + .prepare_node_bundle(&f.node, &frames, &[assigned], NOW) + .await + .unwrap(); + let (node, mut proofs) = f + .directory + .select_node_bundle(&f.node, &proposal, &f.lease, Limits::default(), NOW) + .await + .unwrap(); + f.node = node; + proofs.pop().unwrap() + }}; + } + let first = select!(2); + let old_receipt = select!(3); + let mut publisher = f.publisher(&cell); + let root = publisher.materialize_bundle(&first).await.unwrap(); + let checkpoint = first.materialized_prefix(root, None).unwrap(); + f.node = f + .directory + .checkpoint_bundle_cell(&f.node, &cell.authority, &first, Limits::default(), NOW) + .await + .unwrap(); + let intermediate_receipt = select!(4); + intermediate_receipt + .continues_selected_prefix(&old_receipt, Some(&checkpoint)) + .unwrap(); + // The materializer still owns an older receipt, while new captures have + // already been selected against the first checkpoint's base. + let next_root = publisher.materialize_bundle(&old_receipt).await.unwrap(); + assert!( + old_receipt + .materialized_prefix(root, Some(&checkpoint)) + .is_err() + ); + let next_checkpoint = old_receipt + .materialized_prefix(next_root, Some(&checkpoint)) + .unwrap(); + f.node = f + .directory + .checkpoint_bundle_cell( + &f.node, + &cell.authority, + &old_receipt, + Limits::default(), + NOW, + ) + .await + .unwrap(); + let latest = select!(5); + latest + .continues_selected_prefix(&intermediate_receipt, Some(&next_checkpoint)) + .unwrap(); + let third_root = publisher + .materialize_bundle(&intermediate_receipt) + .await + .unwrap(); + let third_checkpoint = intermediate_receipt + .materialized_prefix(third_root, Some(&next_checkpoint)) + .unwrap(); + f.node = f + .directory + .checkpoint_bundle_cell( + &f.node, + &cell.authority, + &intermediate_receipt, + Limits::default(), + NOW, + ) + .await + .unwrap(); + let newest = select!(6); + newest + .continues_selected_prefix(&latest, Some(&third_checkpoint)) + .unwrap(); + assert!( + newest + .continues_selected_prefix(&latest, Some(&next_checkpoint)) + .is_err() + ); + assert!( + third_checkpoint.retained_bytes() <= MaterializedBundlePrefix::maximum_retained_bytes() + ); +} + #[tokio::test] async fn checkpoint_uses_the_exact_materialized_proof_without_scanning_siblings_or_old_frames() { let mut f = Fixture::new().await; @@ -26,11 +119,13 @@ async fn checkpoint_uses_the_exact_materialized_proof_without_scanning_siblings_ .prepare_node_bundle(&f.node, &frames, &[assigned], NOW) .await .unwrap(); - let (node, _) = f + let (node, mut extended_proofs) = f .directory .select_node_bundle(&f.node, &proposal, &f.lease, Limits::default(), NOW) .await .unwrap(); + let extended = extended_proofs.pop().unwrap(); + extended.continues_selected_prefix(&old, None).unwrap(); f.node = node; f.count.reset(); let checkpoint = f @@ -70,7 +165,7 @@ async fn checkpoint_uses_the_exact_materialized_proof_without_scanning_siblings_ .await .unwrap() .unwrap(); - let suffix = f + let mut suffix = f .directory .load_bundle_coverage(&cell.authority, ¤t, Limits::default()) .await @@ -78,6 +173,44 @@ async fn checkpoint_uses_the_exact_materialized_proof_without_scanning_siblings_ assert_eq!(suffix.base().unwrap(), materialized); assert_eq!(suffix.commit_sequence(), 3); assert_eq!(suffix.locator_count(), frames.len()); + let confirmed = old.materialized_prefix(materialized, None).unwrap(); + assert!(suffix.continues_selected_prefix(&extended, None).is_err()); + suffix + .continues_selected_prefix(&extended, Some(&confirmed)) + .unwrap(); + let mut foreign_root = materialized; + foreign_root.digest[0] ^= 1; + let foreign = old.materialized_prefix(foreign_root, None).unwrap(); + assert!( + suffix + .continues_selected_prefix(&extended, Some(&foreign)) + .is_err() + ); + let retained = suffix.binding.locators.clone(); + suffix.binding.locators.clear(); + assert!( + suffix + .continues_selected_prefix(&extended, Some(&confirmed)) + .is_err(), + "the confirmed root cannot erase its later selected suffix" + ); + suffix.binding.locators = retained.clone(); + suffix.binding.locators[0].frame_digest = Digest::from_bytes([7; 32]); + assert!( + suffix + .continues_selected_prefix(&extended, Some(&confirmed)) + .is_err(), + "retained debt must be byte-identical" + ); + suffix.binding.locators = retained; + let mut changed_prefix = extended; + changed_prefix.binding.locators[0].frame_digest = Digest::from_bytes([6; 32]); + assert!( + suffix + .continues_selected_prefix(&changed_prefix, Some(&confirmed)) + .is_err(), + "the materialized original prefix must match its witness" + ); } #[tokio::test] diff --git a/crates/cellule-runtime/src/node/bundle/tests/managed.rs b/crates/cellule-runtime/src/node/bundle/tests/managed.rs index bdbd506c..e4167e23 100644 --- a/crates/cellule-runtime/src/node/bundle/tests/managed.rs +++ b/crates/cellule-runtime/src/node/bundle/tests/managed.rs @@ -167,20 +167,30 @@ async fn producer_failure_fences_new_work_and_join_preserves_its_cause() { #[tokio::test(flavor = "multi_thread")] async fn managed_producer_selects_actor_prefixes_and_joins_checkpoints_complete_close_and_cold_results() { - managed_actor_case(10, 32 << 20, false).await; + managed_actor_case(10, 32 << 20, false, false).await; } #[tokio::test(flavor = "multi_thread")] async fn managed_actor_retires_215_grouped_commands_per_cell_before_joined_root_materialization() { - managed_actor_case(215, 64 << 20, true).await; + managed_actor_case(215, 64 << 20, true, false).await; } #[tokio::test(flavor = "multi_thread")] async fn managed_actor_retires_65_sequential_captures_before_joined_root_materialization() { - managed_actor_case(65, 64 << 20, false).await; + managed_actor_case(65, 64 << 20, false, false).await; } -async fn managed_actor_case(per_cell: u16, retained_bytes: usize, grouped: bool) { +#[tokio::test(flavor = "multi_thread")] +async fn managed_actor_continues_selected_receipts_across_an_active_root_checkpoint() { + managed_actor_case(215, 64 << 20, true, true).await; +} + +async fn managed_actor_case( + per_cell: u16, + retained_bytes: usize, + grouped: bool, + checkpoint_continuation: bool, +) { let mut f = Fixture::new().await; super::coverage::enroll(&mut f).await; let authority = Arc::new(Authority { @@ -275,7 +285,7 @@ async fn managed_actor_case(per_cell: u16, retained_bytes: usize, grouped: bool) } // Ordinary per-Cell roots are unavailable. The producer must select exact // origin coverage to grant any command ACK, read or retry visibility. - let preparation = dirty.acquire_owned().await.unwrap(); + let mut preparation = Some(dirty.acquire_owned().await.unwrap()); let digest = Digest::from_bytes([9; 32]); let mut commands = tokio::task::JoinSet::new(); let mut results = Vec::new(); @@ -409,13 +419,105 @@ async fn managed_actor_case(per_cell: u16, retained_bytes: usize, grouped: bool) .count(), usize::from(per_cell) * 2 ); + if checkpoint_continuation { + // These captures are selected against the old base while the actor's + // first materializer owns the publisher and waits for preparation. + for byte in per_cell * 2 + 1..=per_cell * 2 + 4 { + renew_actor_lease(&authority, &f.lease).await; + let handle = &cells[usize::from(byte % 2)].3; + let mut request = [0; 16]; + request[..2].copy_from_slice(&byte.to_le_bytes()); + let identity = MutationIdentity { + request_id: RequestId::from_bytes(request), + issued_at_ms: 10, + expires_at_ms: 10_000, + }; + let result = handle + .execute(identity, digest, 20, 1024, 1024, |tx| { + tx.execute("UPDATE counter SET value=value+1", [])?; + Ok(HandlerOutcome::Success(b"managed".to_vec())) + }) + .await + .unwrap(); + results.push((byte, identity, result)); + } + drop(preparation.take()); + tokio::time::timeout(std::time::Duration::from_secs(5), async { + for (target, cell_authority, _, _) in &cells { + loop { + let control = cell_authority + .load(target.cell_id()) + .await + .unwrap() + .unwrap(); + if control.value().ltx_root().unwrap().commit_sequence >= u64::from(per_cell) + && pool.pending(target.cell_id()).await.unwrap().is_none() + { + break; + } + tokio::task::yield_now().await; + } + } + }) + .await + .unwrap(); + // The old receipts are now retired. New proofs use the checkpointed + // base and must preserve their exact retained suffix rather than fence. + for byte in per_cell * 2 + 5..=per_cell * 2 + 6 { + renew_actor_lease(&authority, &f.lease).await; + let handle = &cells[usize::from(byte % 2)].3; + let mut request = [0; 16]; + request[..2].copy_from_slice(&byte.to_le_bytes()); + let identity = MutationIdentity { + request_id: RequestId::from_bytes(request), + issued_at_ms: 10, + expires_at_ms: 10_000, + }; + let result = handle + .execute(identity, digest, 20, 1024, 1024, |tx| { + tx.execute("UPDATE counter SET value=value+1", [])?; + Ok(HandlerOutcome::Success(b"managed".to_vec())) + }) + .await + .unwrap(); + results.push((byte, identity, result)); + } + for (byte, identity, result) in &results { + let handle = &cells[usize::from(byte % 2)].3; + assert_eq!( + handle + .execute(*identity, digest, 21, 1024, 1024, |_| panic!( + "checkpointed retry must not execute again" + )) + .await + .unwrap(), + *result + ); + } + for (_, _, _, handle) in &cells { + assert_eq!( + handle + .query(1024, 1024, |connection| Ok(connection + .query_row("SELECT value FROM counter", [], |row| row.get::<_, i64>(0))? + .to_le_bytes() + .to_vec())) + .await + .unwrap(), + i64::from(per_cell + 3).to_le_bytes() + ); + } + selected_frames += 6; + } + let expected_per_cell = per_cell + if checkpoint_continuation { 3 } else { 0 }; let shutdown = runtime.shutdown(); tokio::pin!(shutdown); - assert!( - tokio::time::timeout(std::time::Duration::from_millis(50), &mut shutdown) - .await - .is_err() - ); + if preparation.is_some() { + assert!( + tokio::time::timeout(std::time::Duration::from_millis(50), &mut shutdown) + .await + .is_err() + ); + } drop(preparation); peers.released.store(true, Ordering::Release); peers.changed.notify_waiters(); @@ -432,7 +534,7 @@ async fn managed_actor_case(per_cell: u16, retained_bytes: usize, grouped: bool) assert_eq!(control.value().state, ControlState::Idle); assert!(control.value().bundle_binding.is_none()); let root = control.value().ltx_root().unwrap(); - assert_eq!(root.commit_sequence, u64::from(per_cell)); + assert_eq!(root.commit_sequence, u64::from(expected_per_cell)); let path = f .scratch .path() @@ -448,7 +550,7 @@ async fn managed_actor_case(per_cell: u16, retained_bytes: usize, grouped: bool) assert_eq!( cold.query_row("SELECT value FROM counter", [], |row| row.get::<_, i64>(0)) .unwrap(), - i64::from(per_cell) + i64::from(expected_per_cell) ); assert_eq!( cold.query_row( @@ -457,7 +559,7 @@ async fn managed_actor_case(per_cell: u16, retained_bytes: usize, grouped: bool) |row| row.get::<_, i64>(0) ) .unwrap(), - i64::from(per_cell) + i64::from(expected_per_cell) ); } assert_eq!( diff --git a/docs/bundle-coverage-implementation.md b/docs/bundle-coverage-implementation.md index af25a6fa..72a4ee73 100644 --- a/docs/bundle-coverage-implementation.md +++ b/docs/bundle-coverage-implementation.md @@ -388,6 +388,12 @@ retries retain the existing grace; shutdown joins accepted materializers and complete original issued-range closure. The manual-feed/unleased fallback retains its bounded 100-ms selection opportunity and ownership of files already preparing. +Confirmed roots retain bounded process-local hash-chain witnesses for original +and intermediate bases. A later selected proof may drop only that identical +materialized prefix and must retain its complete unresolved suffix. Witnesses +expire beyond the 256-locator bound; their heap capacity transfers from memory +admitted before root I/O. They grant no live ACK or origin availability. + The real actor test blocks root preparation until selection, then pauses root CAS and verifies that every selected capture file is absent. It issues a later write and verifies that an unproven suffix refuses a query before its handler From a52fe28542f2839b30b077ca775adba7fe8b1809 Mon Sep 17 00:00:00 2001 From: forhappy Date: Thu, 8 Oct 2026 18:02:29 -0700 Subject: [PATCH 046/102] Record failed checkpoint-continuity performance qualification --- .../docs/write-performance-design.md | 15 +- docs/bundle-coverage-implementation.md | 10 +- docs/pr67-async-root-measurement.md | 4 + .../pr67-checkpoint-continuity-measurement.md | 151 ++++++++++++++++++ docs/write-performance-delivery.md | 12 +- 5 files changed, 185 insertions(+), 7 deletions(-) create mode 100644 docs/pr67-checkpoint-continuity-measurement.md diff --git a/crates/cellule-runtime/docs/write-performance-design.md b/crates/cellule-runtime/docs/write-performance-design.md index 3da1b9eb..c5678a1f 100644 --- a/crates/cellule-runtime/docs/write-performance-design.md +++ b/crates/cellule-runtime/docs/write-performance-design.md @@ -98,11 +98,16 @@ preparing. Process-local hash-chain witnesses retain no frame bodies and expire when their original prefix exceeds the 256-locator proof bound. Their memory transfers from admission acquired before root I/O and releases with the worker. Regression cases exercise live writes, read/retry visibility, joined cold -restore and successive intermediate bases. This correctness change still needs -an end-to-end measurement; it supplies no new durability or origin-availability -proof. - -The latest [asynchronous-root comparison](../../../docs/pr67-async-root-measurement.md) +restore and successive intermediate bases. This supplies no new durability or +origin-availability proof. The +[checkpoint-continuity measurement](../../../docs/pr67-checkpoint-continuity-measurement.md) +at `4a8f55c` completes 185.83 Fleet writes/s versus 164.73 but still returns +304,151 measured errors and fails warm ACK availability. Bucket completes +215.08 writes/s versus 227.15 with passing audits and higher p99. All contributor +checks pass, but the original application load failure persists. No acceptable +improvement or parity is established. + +The earlier [asynchronous-root comparison](../../../docs/pr67-async-root-measurement.md) at `6d62d41` completes 183.65 Fleet writes/s versus 115.27, with successful scheduled p99 of 1,782.74 ms. It returns 297,811 measured errors and fails its warm ACK audit. No cold or successful drain evidence follows. Root density is diff --git a/docs/bundle-coverage-implementation.md b/docs/bundle-coverage-implementation.md index 72a4ee73..54804d57 100644 --- a/docs/bundle-coverage-implementation.md +++ b/docs/bundle-coverage-implementation.md @@ -11,7 +11,15 @@ Fleet writes/s from 525.67 and failed availability/drain. The subsequent [coverage-race measurement](pr67-coverage-race-measurement.md) at `e40ecd6` passes all-ACK warm/cold audit and joined drain, but completes 100.20 Fleet writes/s versus 106.05 before the fix: no demonstrated throughput gain. -The latest [asynchronous-root comparison](pr67-async-root-measurement.md) at +The latest [checkpoint-continuity comparison](pr67-checkpoint-continuity-measurement.md) +at `4a8f55c` verifies live writes across a confirmed root and successive +intermediate-base receipts, with bounded, admitted prefix witnesses. It measures +185.83 Fleet writes/s versus 164.73, but still returns 304,151 errors and fails +warm ACK availability. Bucket completes 215.08 writes/s versus 227.15 with +passing audits and higher p99. All contributor checks pass; no acceptable +performance improvement or parity is established. PR #67 remains a draft. + +The earlier [asynchronous-root comparison](pr67-async-root-measurement.md) at `6d62d41` retires exact selected captures and schedules admitted roots separately. It completes 183.65 Fleet writes/s versus 115.27 before, but returns 297,811 measured request errors and fails its warm ACK audit. Cold recovery and successful diff --git a/docs/pr67-async-root-measurement.md b/docs/pr67-async-root-measurement.md index 155b2193..61528d05 100644 --- a/docs/pr67-async-root-measurement.md +++ b/docs/pr67-async-root-measurement.md @@ -1,5 +1,9 @@ # Asynchronous roots: measured Fleet availability regression +The newer [checkpoint-continuity measurement](pr67-checkpoint-continuity-measurement.md) +measures `4a8f55c` against this release. Its focused regressions pass, but Fleet +availability still fails and Bucket completion rate falls in that pair. + **Measured candidate: 183.65 Fleet writes/s and 204.10 Bucket writes/s.** The Fleet candidate returns **297,811 measured request errors**, versus zero before the change, and fails its warm ACK audit. Its higher completion rate and diff --git a/docs/pr67-checkpoint-continuity-measurement.md b/docs/pr67-checkpoint-continuity-measurement.md new file mode 100644 index 00000000..d632e6ef --- /dev/null +++ b/docs/pr67-checkpoint-continuity-measurement.md @@ -0,0 +1,151 @@ +# Checkpoint continuity: measured write results + +**Current measured completion rates: 185.83 Fleet writes/s and 215.08 Bucket +writes/s. No acceptable performance improvement or celld parity is established.** +Fleet still returns 304,151 measured errors and fails its warm ACK audit. Bucket +has zero request errors and passing ACK audits, but completes 5.3% fewer writes +than the immediate baseline and has worse successful-write p99. Every point +fails qualification. PR #67 remains a draft. + +## Tested change and provenance + +Candidate: `4a8f55cc99414e6a3907531c774a184842e1de3e`. Immediate functional +baseline: `6d62d4117712877f2c250d58f9b1cd60b529819b`, the candidate in the +[previous asynchronous-root measurement](pr67-async-root-measurement.md). +This comparison is against that release, not a fresh origin/main build. +Celld is unchanged at `f2bf648663a610eefde71f3547ad61e9b896b1f0` and its pinned +image. Both Cellule binaries come from committed source. + +The follow-up fixes a reproduced selected-prefix continuity failure across +confirmed asynchronous checkpoints. It preserves receipts selected against an +intermediate base while an older root prepares. Process-local hash-chain +witnesses admit only removal of the identical materialized prefix and require +the complete remaining suffix. They retain no frame bodies, expire beyond the +256-locator bound, and transfer heap admission acquired before root I/O to the +worker. They grant no new ACK, authority or origin-availability proof. + +Before the fix, the real actor checkpoint case failed twice; a successive-base +case also failed. Both now pass, including live writes, read/retry visibility, +joined shutdown and cold SQLite checks. All 81 bundle tests pass. On the final +frozen snapshot, all 11 contributor verification routes pass with **1,961 tests +passed, zero failed and 38 ignored**. An earlier broad run failed an unchanged +peer-HTTP concurrency test; its isolated rerun and the final full suite pass. +These results do not remove the measured application availability failure below. + +The Linux build's 1,905-file source manifest matches the frozen verification +snapshot exactly. The load generator and auditor are byte-identical across +arms; comparison verifies fixture, runner, image and resource provenance. + +| Artifact | SHA-256 | +| --- | --- | +| Frozen/framework source manifest | `4a48fac79be76d9aa037090a3a7b64a16ff81ac1412d07ae83f548df920f52a4` | +| Adapted Linux build source | `0503aa0800940982a15a79fdc96dc68113d85f14f480f2c2376fe2ccd1b24430` | +| Candidate SQL binary | `4c7874206d84885064595e04a8759075ea45ce44bc90b4eb02bfae97fa945c13` | +| Common load generator | `417f07b0424d27df75b1dca22666a7adb621db3cd11fdb1048eb9247a664e60b` | +| Common auditor | `6595c24b0be217e181e8a9aa7de5cf478669ac341b37b76754fdcf0b3865b7b2` | + +## Workload and limits + +Six fresh cases ran sequentially: before/after/celld in Fleet, then Bucket. +Each uses 1,000 uniform Cells, 96-byte values, INSERT plus SELECT per command, +a two-hour durable retry/result ledger, 128 clients and 128 queue slots. +Warmup is 30 seconds and the measured window 60 seconds, with one repetition. +Fleet offers 15,000 writes/s with two followers; Bucket offers 2,000/s. + +The ARM64 Linux VM has **8 CPUs and 8 GiB total shared memory**. Serving +containers retain an 8-CPU/16-GiB ceiling and 4-GiB tmpfs; the client has a +4-CPU/4-GiB ceiling and RustFS 2 CPUs/8 GiB. Ceilings exceed VM resources. +The unchanged external provider adaptation, fresh Linux volumes, 64-MiB Cellule +retention budget, 1-GiB managed disk budget and original acceptance gates remain. +No builds or contributor checks overlap the timed windows. + +This is a constrained, overloaded SQL application diagnostic. It is not the +bounded KV laptop workload, a dedicated 8-vCPU/16-GiB serving node, a 2,000-Cell +qualification, or the required three paired five-minute repetitions. tmpfs +does not qualify physical-device durability. + +## Reconciled measurements + +TPS counts successful logical writes completed inside the measured window. +Successful p99 is independently replayed from journals, includes trailing +successes, and excludes errors and drops. Scheduled latency starts at offered +arrival; request latency starts at issuance. The qualification reporter also +checks all-attempt latency. Its percentile is not interchangeable with the +successful-write percentile below. + +| Mode / system | Successful writes/s | Successful scheduled p99 ms | Successful request p99 ms | Measured errors | Queue drops | +| --- | ---: | ---: | ---: | ---: | ---: | +| Fleet / baseline | 164.73 | 2,636.38 | 1,718.73 | 312,664 | 577,381 | +| Fleet / candidate | 185.83 | 2,002.08 | 1,646.30 | 304,151 | 584,699 | +| Fleet / celld | 4,041.48 | 255.24 | 122.96 | 1,948 | 655,524 | +| Bucket / baseline | 227.15 | 5,289.89 | 3,522.02 | 0 | 106,115 | +| Bucket / candidate | 215.08 | 6,146.30 | 5,376.70 | 0 | 106,839 | +| Bucket / celld | 391.45 | 3,768.85 | 2,120.32 | 210 | 96,257 | + +All six cases have zero warmup request errors, but drop warmup offers. +Independent replay reconciles all 900,000 Fleet or 120,000 Bucket offers to +attempts or drops, every attempt to success/error, window/trailing completions, +and complete ACK cohort counts and digests. **These rates are overloaded +completion counts, not sustainable capacities.** + +Fleet completes 12.8% more writes in this pair, but its errors and failed +availability audit prevent an acceptable-gain claim. Lower survivor p99 does +not establish overall improvement. Bucket completes 5.3% fewer writes with +16.2% higher successful scheduled p99. Its fixture bypasses the managed bundle +producer, so these single-pair differences cannot be attributed to the fix. +Celld's owner failures below also prevent a qualified reference-rate claim. +No repeatable improvement, read-capacity result or parity is established. + +## Availability, recovery and lifecycle + +| Mode / system | Complete ACK cohort | Warm audit errors / retry checks | Cold read/retry | Recorded successful drain seconds | +| --- | ---: | --- | --- | ---: | +| Fleet / baseline | 20,027 | 14,689 / 5,338 | not reached | absent | +| Fleet / candidate | 21,678 | 14,681 / 6,997 | not reached | absent | +| Fleet / celld | 464,260 | 463,665 / 595 | not reached | absent | +| Bucket / baseline | 25,134 | 0 / 25,134 | pass: all 25,134 | 5.98 | +| Bucket / candidate | 23,835 | 0 / 23,835 | pass: all 23,835 | 8.51 | +| Bucket / celld | 44,814 | 44,814 / 0 | not reached | absent | + +Both Cellule Fleet audits report HTTP 503 failures. The candidate owner exits +zero during cleanup, but the failed case has no successful aggregate drain +record or cold audit; its exit alone cannot supply them. Celld Fleet's original +owner is **OOMKilled=true, exit 137**. Celld Bucket's owner is **OOMKilled=false, +exit 3**; its log records ambiguous renewals and node-lease watchdog self-fencing. +No case is omitted or silently retried. These failures establish unavailable +audits, not proven mutation loss. + +Failed cases lack required later provider filesystem/lifecycle observations. +Absence of those records does not prove provider failure and cannot pass the +provider health gate. Passing Bucket cold checks follow graceful joined drain; +they do not qualify recovery of a failed owner's entire Fleet-ACK suffix. + +## Remaining measured bottleneck + +The canonical Fleet window-cost report shows PUTs per completed command moving +from **0.5083 to 0.4697**, GET/range attempts from **13.0395 to 13.8594**, and +materialized commands per root from **11.30 to 11.70**. Steady Bundle response +counts remain zero. These are storage API observations; SDK-internal retry +attempts are not separately measured. + +Candidate retained memory grows from **28.41 to 59.52 MiB of 64 MiB**, accounted +unpublished node-log bytes grow, and oldest publication reaches **47,453 ms**. +Two boundary samples do not prove bounded debt. The checkpoint density target +is 215 commands and the conditional PUT target 0.05 per command; neither is met. +The observed application error's precise cause remains unisolated. Fixing the +focused continuity cases has not fixed the original load-test failure. + +Next work must isolate application refusal/read visibility and resource +admission under this reproduced load, establish admitted materializer progress, +connect Bucket to shared selection, and reduce complete verification/publication +cost. Successful all-ACK availability/drain, sustainable A/A capacity, read-only +and mixed guardrails, complete failed-owner recovery, safe collection and the +unchanged qualification gates remain open. + +Raw journals, failures, manifests, binaries, comparisons and the immutable +evidence index remain outside Git under +`/Volumes/Workspace/crabbuild-target/cellule-write-perf-8ad1/`, with prefix +`checkpoint-transition-20261008`. Only this compact report belongs in the repo. +The version-2 index covers **17,146 files / 1,687,273,628 bytes**; every entry +was rehashed and matched. Its SHA-256 is +`9566c2896270f252d5636a0bd975ffe96e625e8853632bf0f90654eec2550934`. diff --git a/docs/write-performance-delivery.md b/docs/write-performance-delivery.md index d5aee747..d7228aad 100644 --- a/docs/write-performance-delivery.md +++ b/docs/write-performance-delivery.md @@ -6,7 +6,17 @@ comparison retains exact retry and cold-state audits. **Celld write parity has not been established.** The [proposal](write-performance-proposal.md) remains the acceptance contract; completing tests or a load run does not pass its gates. -The latest [asynchronous-root measurement](pr67-async-root-measurement.md) +The latest [checkpoint-continuity measurement](pr67-checkpoint-continuity-measurement.md) +records runtime commit `4a8f55c`: **185.83 Fleet writes/s and 215.08 Bucket +writes/s**, versus 164.73 and 227.15 for its immediate `6d62d41` baseline. +Fleet still returns 304,151 measured errors and fails its warm ACK audit. +Bucket has passing ACK audits but a 5.3% lower completion rate and higher p99. +Focused checkpoint regressions and all contributor checks pass; the original +application load failure persists. Celld also fails these fresh Fleet/Bucket +cases through OOM/self-fencing. Every point fails qualification; no acceptable +improvement or parity is established, and PR #67 remains a draft. + +The earlier [asynchronous-root measurement](pr67-async-root-measurement.md) records runtime commit `6d62d41`: 183.65 Fleet writes/s and 204.10 Bucket writes/s versus 115.27 and 196.27 in fresh paired windows. Fleet successful scheduled p99 is 1,782.74 ms, but the candidate returns 297,811 measured errors and fails its From 524635f15491949712470e43da02c86fc47a5b9e Mon Sep 17 00:00:00 2001 From: forhappy Date: Thu, 8 Oct 2026 18:36:51 -0700 Subject: [PATCH 047/102] docs(perf): record honest release repeat and selection deadline failures --- .../docs/write-performance-design.md | 7 +- docs/bundle-coverage-implementation.md | 7 +- .../pr67-checkpoint-continuity-measurement.md | 12 +- docs/pr67-release-repeat-measurement.md | 126 ++++++++++++++++++ docs/write-performance-delivery.md | 12 +- 5 files changed, 158 insertions(+), 6 deletions(-) create mode 100644 docs/pr67-release-repeat-measurement.md diff --git a/crates/cellule-runtime/docs/write-performance-design.md b/crates/cellule-runtime/docs/write-performance-design.md index c5678a1f..7fbbd3c2 100644 --- a/crates/cellule-runtime/docs/write-performance-design.md +++ b/crates/cellule-runtime/docs/write-performance-design.md @@ -105,7 +105,12 @@ at `4a8f55c` completes 185.83 Fleet writes/s versus 164.73 but still returns 304,151 measured errors and fails warm ACK availability. Bucket completes 215.08 writes/s versus 227.15 with passing audits and higher p99. All contributor checks pass, but the original application load failure persists. No acceptable -improvement or parity is established. +improvement or parity is established. A subsequent +[release-build repeat](../../../docs/pr67-release-repeat-measurement.md) of the +same binary completes 107.95 Fleet and 268.12 Bucket writes/s. Fleet still fails +warm ACK availability; Bucket audits pass but delivery targets fail. Diagnostic +logs identify shared-selection deadlines that fence Cells and publication +backlog refusals. This supplies no demonstrated throughput improvement. The earlier [asynchronous-root comparison](../../../docs/pr67-async-root-measurement.md) at `6d62d41` completes 183.65 Fleet writes/s versus 115.27, with successful diff --git a/docs/bundle-coverage-implementation.md b/docs/bundle-coverage-implementation.md index 54804d57..470d8d9f 100644 --- a/docs/bundle-coverage-implementation.md +++ b/docs/bundle-coverage-implementation.md @@ -11,7 +11,12 @@ Fleet writes/s from 525.67 and failed availability/drain. The subsequent [coverage-race measurement](pr67-coverage-race-measurement.md) at `e40ecd6` passes all-ACK warm/cold audit and joined drain, but completes 100.20 Fleet writes/s versus 106.05 before the fix: no demonstrated throughput gain. -The latest [checkpoint-continuity comparison](pr67-checkpoint-continuity-measurement.md) +The latest [release-build repeat](pr67-release-repeat-measurement.md) of the same +`4a8f55c` binary completes 107.95 Fleet and 268.12 Bucket writes/s. Fleet still +fails warm ACK availability; Bucket audits pass but delivery targets fail. +Diagnostic logs identify shared-selection deadlines that fence Cells and +publication backlog refusals. No acceptable improvement or parity is established. +The earlier [checkpoint-continuity comparison](pr67-checkpoint-continuity-measurement.md) at `4a8f55c` verifies live writes across a confirmed root and successive intermediate-base receipts, with bounded, admitted prefix witnesses. It measures 185.83 Fleet writes/s versus 164.73, but still returns 304,151 errors and fails diff --git a/docs/pr67-checkpoint-continuity-measurement.md b/docs/pr67-checkpoint-continuity-measurement.md index d632e6ef..1f7cf288 100644 --- a/docs/pr67-checkpoint-continuity-measurement.md +++ b/docs/pr67-checkpoint-continuity-measurement.md @@ -1,6 +1,10 @@ # Checkpoint continuity: measured write results -**Current measured completion rates: 185.83 Fleet writes/s and 215.08 Bucket +The later [release repeat](pr67-release-repeat-measurement.md) of the same +binary completes 107.95 Fleet and 268.12 Bucket writes/s. It retains failed +cases and isolates a shared-selection deadline failure path. + +**This comparison's completion rates: 185.83 Fleet writes/s and 215.08 Bucket writes/s. No acceptable performance improvement or celld parity is established.** Fleet still returns 304,151 measured errors and fails its warm ACK audit. Bucket has zero request errors and passing ACK audits, but completes 5.3% fewer writes @@ -132,8 +136,10 @@ Candidate retained memory grows from **28.41 to 59.52 MiB of 64 MiB**, accounted unpublished node-log bytes grow, and oldest publication reaches **47,453 ms**. Two boundary samples do not prove bounded debt. The checkpoint density target is 215 commands and the conditional PUT target 0.05 per command; neither is met. -The observed application error's precise cause remains unisolated. Fixing the -focused continuity cases has not fixed the original load-test failure. +The subsequent [diagnostic and release repeat](pr67-release-repeat-measurement.md) +records selection deadlines that fence Cells and sampled pre-SQL publication +backlog refusals. Fixing the focused continuity cases has not fixed the +original load-test failure. Next work must isolate application refusal/read visibility and resource admission under this reproduced load, establish admitted materializer progress, diff --git a/docs/pr67-release-repeat-measurement.md b/docs/pr67-release-repeat-measurement.md new file mode 100644 index 00000000..2a96cf0b --- /dev/null +++ b/docs/pr67-release-repeat-measurement.md @@ -0,0 +1,126 @@ +# PR 67: release-build write measurement repeat + +**Latest completion rates: 107.95 Fleet writes/s and 268.12 Bucket writes/s. +No acceptable performance improvement or celld parity is established.** Fleet +fails request delivery and warm ACK availability. Bucket passes its ACK audits +but misses throughput and latency targets. PR #67 remains a draft. + +## Source and workload + +The branch head at measurement was `a52fe28542f2839b30b077ca775adba7fe8b1809`. +Its differences from functional commit +`4a8f55cc99414e6a3907531c774a184842e1de3e` are documentation only. This repeat +uses that verified release binary, not the separate diagnostic logging build. +Celld remains pinned at `f2bf648663a610eefde71f3547ad61e9b896b1f0` and the +same image. No production changes were made during this measurement. + +Four fresh cases ran sequentially: Cellule Fleet, celld Fleet, Cellule Bucket, +celld Bucket. Each has 1,000 uniformly selected Cells, 96-byte values, SQL +INSERT plus SELECT, a two-hour durable retry/result ledger, 128 clients and +128 queue slots. Each has a 30-second warmup and 60-second measured window. +Fleet offers 15,000 writes/s with two followers; Bucket offers 2,000/s. +Prefixes and provider volumes are fresh. No build or contributor suite overlaps +the timed windows. The owned Docker context is `colima-c67r2`. + +The ARM64 Linux VM has **8 CPUs and 8 GiB total shared memory**. Serving +containers have 8-CPU/16-GiB ceilings and 4-GiB tmpfs; the client has a +4-CPU/4-GiB ceiling and RustFS 2 CPUs/8 GiB. These ceilings exceed VM resources. +The existing external provider adaptation, 64-MiB Cellule retention budget, +1-GiB managed disk budget and acceptance gates remain unchanged. + +This SQL diagnostic does not qualify a dedicated 8-vCPU/16-GiB serving node, +the laptop KV workload, physical-device durability, or the 2,000-Cell goal. +Three paired five-minute repetitions, a sustainable capacity search and read +guardrails remain outstanding. + +## Independently reconciled results + +TPS counts successful logical writes completed inside the measured window. +The successful p99 below is replayed from request journals, includes trailing +successes and excludes errors and dropped offers. Scheduled latency starts at +the offered arrival; request latency starts at issuance. The qualification +report also checks all-attempt latency. These overloaded completion counts +are not sustainable capacities. + +| Mode / system | Successful writes/s | Successful scheduled p99 ms | Successful request p99 ms | Measured errors | Queue drops | +| --- | ---: | ---: | ---: | ---: | ---: | +| Fleet / Cellule | 107.95 | 5,243.30 | 3,137.32 | 499,450 | 394,049 | +| Fleet / celld | 4,470.70 | 158.21 | 70.29 | 2,395 | 629,337 | +| Bucket / Cellule | 268.12 | 4,059.96 | 3,399.88 | 0 | 103,657 | +| Bucket / celld | 1,392.55 | 586.52 | 323.98 | 0 | 36,198 | + +All four warmup request-error counts are zero, but all drop warmup offers. +Independent replay reconciles every 900,000 Fleet or 120,000 Bucket offer, +attempt, success, error, drop and window/trailing completion. Complete ACK +cohort counts and hashes match. Comparisons verify identical fixture, +load-generator, auditor, image, host and runner provenance. Every point fails +qualification; no failed case is omitted or silently retried. + +The [previous measurement](pr67-checkpoint-continuity-measurement.md) of this +same Cellule binary completed 185.83 Fleet and 215.08 Bucket writes/s. The +repeat changes those counts by -41.9% and +24.7%. This variation cannot be +attributed to a new optimization. The earlier before/after pair against +`6d62d41` also failed availability/delivery and did not establish improvement. + +## ACK availability and drain + +| Mode / system | Complete ACK cohort | Warm errors / retry checks | Cold read/retry | Recorded successful drain seconds | +| --- | ---: | --- | --- | ---: | +| Fleet / Cellule | 16,190 | 16,057 / 133 | not reached | absent | +| Fleet / celld | 463,838 | 447,870 / 15,968 | not reached | absent | +| Bucket / Cellule | 24,880 | 0 / 24,880 | pass: all 24,880 | 9.14 | +| Bucket / celld | 129,029 | 0 / 129,029 | pass: all 129,029 | 1.79 | + +Cellule Fleet's warm audit returns HTTP 503 failures. Its owner exits zero +during cleanup, which does not supply a successful aggregate drain or cold +audit. Celld Fleet's owner is OOM-killed, exit 137; its audit records HTTP 500 +failures. These establish unavailable audits, not proven mutation loss. +Required later provider health/lifecycle observations are missing for failed +cases and fail their gates. Passing Bucket cold checks follow graceful drain; +they do not qualify recovery of a failed owner's complete Fleet-ACK suffix. + +## Reproduced failure path and remaining bottleneck + +A separate, explicitly marked diagnostic build records **466 distinct Cells** +with `exact shared selection failed` / `Deadline`, plus 128 sampled HTTP 503 +responses with `NotStarted(Capacity("publication backlog"))`. Its warm audit +fails all 16,059 checked ACKs. Those bounded logging counts identify failure +paths; its timing is not included in the release comparison above. + +The current [selection task](../crates/cellule-runtime/src/cell/actor/materialization/mod.rs) +holds the Cell publisher while awaiting the producer's selected capture prefix +and wraps that wait plus worker cleanup in a ten-second timeout. A timeout +fences the Cell even when a Fleet proof has already granted ACKs. Root dispatch +also requires that publisher, so it cannot start for the Cell during the wait. +Separating selection readiness from admitted cleanup/root ownership needs a +regression with delayed selection, valid Fleet ACKs, read/retry visibility, +bounded capture debt and joined drain. Increasing the timeout alone does not +resolve that coupling. This diagnosis does not establish the sole throughput +bottleneck or demonstrate a fix. + +The release Fleet window records 11.00 materialized commands/root, 0.7403 +PUT attempts/completed write and 13.2234 GET/range attempts/completed write. +These are storage API boundary deltas, including work for earlier pending cuts; +provider SDK internal retries are not separately counted. Steady Bundle ACKs +remain zero. Retention grows from 27.80 to 59.72 MiB of 64 MiB, unpublished +node-log bytes grow, and oldest publication reaches 49,784 ms. Two boundary +samples do not establish bounded debt. The 215-command and conditional +0.05-PUT targets remain unmet. Bucket wiring still bypasses the managed producer. + +## Evidence and verification + +Production source is unchanged from the frozen snapshot whose eleven +contributor routes passed: 1,961 tests passed, zero failed and 38 ignored. +Those checks do not establish application throughput or availability. + +Raw case journals, complete ACK cohorts, provider objects/logs, failures, +binaries, manifests and the diagnostic overlay remain outside Git under +`/Volumes/Workspace/crabbuild-target/cellule-write-perf-8ad1`. +The retained repeat/diagnostic evidence index covers 1,559 files and +1,679,263,260 bytes; all entries were rehashed successfully. Index SHA-256: +`3e663130645404ae76d157de12bac01e5180a4a51322957cbfad3ffc90fb3944`. + +The [delivery report](write-performance-delivery.md) and +[runtime design](../crates/cellule-runtime/docs/write-performance-design.md) +retain the remaining availability, shared Bucket selection, publication cost, +failed-owner recovery, safe collection and full qualification requirements. diff --git a/docs/write-performance-delivery.md b/docs/write-performance-delivery.md index d7228aad..3f41805f 100644 --- a/docs/write-performance-delivery.md +++ b/docs/write-performance-delivery.md @@ -6,7 +6,17 @@ comparison retains exact retry and cold-state audits. **Celld write parity has not been established.** The [proposal](write-performance-proposal.md) remains the acceptance contract; completing tests or a load run does not pass its gates. -The latest [checkpoint-continuity measurement](pr67-checkpoint-continuity-measurement.md) +The latest [release-build repeat](pr67-release-repeat-measurement.md) measures +the unchanged `4a8f55c` binary at **107.95 Fleet writes/s and 268.12 Bucket +writes/s**. Fleet returns 499,450 measured errors and fails 16,057 of 16,190 +warm ACK checks. Bucket passes all 24,880 warm/cold checks but drops 103,657 +offers and misses latency/throughput targets. Celld completes 4,470.70 Fleet +and 1,392.55 Bucket writes/s; Fleet is OOM-killed, while Bucket audits pass. +Every point fails qualification. Diagnostic logs identify selection deadlines +that fence Cells and publication backlog refusals. No acceptable improvement +or parity is established; PR #67 remains a draft. + +The earlier [checkpoint-continuity measurement](pr67-checkpoint-continuity-measurement.md) records runtime commit `4a8f55c`: **185.83 Fleet writes/s and 215.08 Bucket writes/s**, versus 164.73 and 227.15 for its immediate `6d62d41` baseline. Fleet still returns 304,151 measured errors and fails its warm ACK audit. From 6c909a6dec983bc222a0e1c047e878808f677cb2 Mon Sep 17 00:00:00 2001 From: forhappy Date: Thu, 8 Oct 2026 18:54:39 -0700 Subject: [PATCH 048/102] fix(runtime): keep the publisher free while bundle selection waits --- .../docs/write-performance-design.md | 10 + .../src/cell/actor/admission.rs | 1 + .../src/cell/actor/inventory/tests.rs | 1 + .../src/cell/actor/materialization/mod.rs | 26 +- .../cell/actor/materialization/readiness.rs | 37 +++ .../src/cell/actor/requests.rs | 9 +- .../cellule-runtime/src/cell/actor/state.rs | 8 + .../src/cell/actor/tasks/activation.rs | 1 + .../src/cell/actor/tasks/mod.rs | 5 + .../src/cell/actor/tasks/publication.rs | 30 ++ .../src/node/bundle/tests/managed.rs | 2 +- .../src/node/bundle/tests/mod.rs | 1 + .../src/node/bundle/tests/readiness.rs | 299 ++++++++++++++++++ .../src/node/log_shipper/publication/mod.rs | 5 + crates/cellule-runtime/src/publication/mod.rs | 8 + 15 files changed, 434 insertions(+), 9 deletions(-) create mode 100644 crates/cellule-runtime/src/cell/actor/materialization/readiness.rs create mode 100644 crates/cellule-runtime/src/node/bundle/tests/readiness.rs diff --git a/crates/cellule-runtime/docs/write-performance-design.md b/crates/cellule-runtime/docs/write-performance-design.md index 7fbbd3c2..88716b82 100644 --- a/crates/cellule-runtime/docs/write-performance-design.md +++ b/crates/cellule-runtime/docs/write-performance-design.md @@ -92,6 +92,16 @@ locator/byte pressure gates new commands. Root age of 45 seconds, drain, migration and fallback also request materialization. Due hints expose the exact selected head while a root lags. +Selection readiness is observed once per Cell without owning its publisher. +The observer shares the original receipt and admitted metadata; it grants no +ACK, narrowed proof or new authority. Only a ready oldest prefix enters exact +cleanup. Its unselected suffix remains in the bounded original queue, allowing +older root debt to prepare. The existing ten-second cleanup timeout begins +after selection readiness, rather than timing an origin wait after Fleet ACKs. +Fencing/removal cancels only observation; original native publication and +checkpoint/drain obligations remain joined. Performance qualification of this +change remains required. + A follow-up preserves exact selected suffixes across a confirmed checkpoint, including receipts selected against intermediate bases while older roots were preparing. Process-local hash-chain witnesses retain no frame bodies and expire diff --git a/crates/cellule-runtime/src/cell/actor/admission.rs b/crates/cellule-runtime/src/cell/actor/admission.rs index 8510e3c6..a001b906 100644 --- a/crates/cellule-runtime/src/cell/actor/admission.rs +++ b/crates/cellule-runtime/src/cell/actor/admission.rs @@ -56,6 +56,7 @@ pub(super) fn finish_migration(active: &mut ActiveCell, fenced: bool) -> Coordin pub(super) fn fence_active(active: &mut ActiveCell) { active.cancel_compaction_admission(); + drop(active.selection_waiter.take()); active.coordination.step(CoordinationInput::Fence); fence_admission(&active.admission); if let Some(transfer) = active.transfer.take() { diff --git a/crates/cellule-runtime/src/cell/actor/inventory/tests.rs b/crates/cellule-runtime/src/cell/actor/inventory/tests.rs index d6082598..d99afe3f 100644 --- a/crates/cellule-runtime/src/cell/actor/inventory/tests.rs +++ b/crates/cellule-runtime/src/cell/actor/inventory/tests.rs @@ -103,6 +103,7 @@ async fn stale_actor_probe_cannot_clear_newer_mutation_markers_or_replace_newer_ durability_submitter: publisher.durability_submitter(), publisher: Some(publisher), publications: VecDeque::new(), + selection_waiter: None, publishing_since: None, root_debt: None, materializing: false, diff --git a/crates/cellule-runtime/src/cell/actor/materialization/mod.rs b/crates/cellule-runtime/src/cell/actor/materialization/mod.rs index 37d847ab..c7733201 100644 --- a/crates/cellule-runtime/src/cell/actor/materialization/mod.rs +++ b/crates/cellule-runtime/src/cell/actor/materialization/mod.rs @@ -2,6 +2,8 @@ use super::*; +mod readiness; + pub(super) const CHECKPOINT_COMMANDS: u64 = 215; const MAX_ROOT_AGE: std::time::Duration = std::time::Duration::from_secs(45); const MAX_MATERIALIZERS: usize = 8; @@ -22,9 +24,27 @@ pub(super) fn start_selection( active: &mut ActiveCell, pool: &SqlWorkerPool, tasks: &mut JoinSet, - publisher: CellPublisher, ) { - let coverage: Vec<_> = active.publications.drain(..).collect(); + let ready = active + .publications + .iter() + .take_while(|queued| { + queued + .durability + .as_ref() + .is_some_and(PendingDurability::selection_ready) + }) + .count(); + if ready == 0 { + readiness::wait(cell, active, tasks); + return; + } + let Some(publisher) = active.publisher.take() else { + return; + }; + // Retire only a selected oldest prefix. The bounded unselected suffix + // stays in the original queue, without excluding root preparation. + let coverage: Vec<_> = active.publications.drain(..ready).collect(); let covered = coverage.len() as u64; let retained_bytes = coverage .iter() @@ -34,7 +54,7 @@ pub(super) fn start_selection( let effect_id = active.begin_task(CoordinationEffect::Publication); active.publishing_since = coverage.first().map(|queued| queued.submitted_at); // Worker cleanup is dispatched through the same owned pool as SQL. The - // publisher token excludes root preparation throughout exact selection. + // publisher token excludes root preparation only during verified cleanup. let pool = pool.clone(); tasks.spawn(async move { let result = async { diff --git a/crates/cellule-runtime/src/cell/actor/materialization/readiness.rs b/crates/cellule-runtime/src/cell/actor/materialization/readiness.rs new file mode 100644 index 00000000..64861f09 --- /dev/null +++ b/crates/cellule-runtime/src/cell/actor/materialization/readiness.rs @@ -0,0 +1,37 @@ +//! Observe original selection without owning the Cell publication token. +use super::*; + +pub(super) fn wait(cell: CellId, active: &mut ActiveCell, tasks: &mut JoinSet) { + if active.selection_waiter.is_some() || active.coordination.is_fenced() { + return; + } + let Some(durability) = active + .publications + .front() + .and_then(|queued| queued.durability.as_ref()) + .cloned() + else { + return; + }; + let generation = active.generation; + let cancelled = tokio_util::sync::CancellationToken::new(); + active.selection_waiter = Some(cancelled.clone().drop_guard()); + // Resident Cell admission covers one bounded observer. It shares the + // existing capture/selected metadata credits and creates no new proof. + // Cancellation stops only observation; the original producer still owns + // every issued cut and its checkpoint/drain obligations. + tasks.spawn(async move { + let result = tokio::select! { + selected = durability.selected_capture() => selected.and_then(|capture| { + capture.ok_or(Error::Control("managed selection lacks capture"))?; + Ok(()) + }), + () = cancelled.cancelled() => Ok(()), + }; + TaskResult::BundleSelectionReady { + cell, + generation, + result, + } + }); +} diff --git a/crates/cellule-runtime/src/cell/actor/requests.rs b/crates/cellule-runtime/src/cell/actor/requests.rs index 2f06bcc1..455d3e66 100644 --- a/crates/cellule-runtime/src/cell/actor/requests.rs +++ b/crates/cellule-runtime/src/cell/actor/requests.rs @@ -395,11 +395,7 @@ pub(super) fn start_publication( pool: &SqlWorkerPool, tasks: &mut JoinSet, ) { - let Some(mut publisher) = active.publisher.take() else { - return; - }; if active.publications.is_empty() { - active.publisher = Some(publisher); return; } if active.publications.iter().all(|queued| { @@ -408,9 +404,12 @@ pub(super) fn start_publication( .as_ref() .is_some_and(PendingDurability::has_managed_bundle_capture) }) { - super::materialization::start_selection(cell, active, pool, tasks, publisher); + super::materialization::start_selection(cell, active, pool, tasks); return; } + let Some(mut publisher) = active.publisher.take() else { + return; + }; if active.root_debt.is_some() { active.publisher = Some(publisher); return; diff --git a/crates/cellule-runtime/src/cell/actor/state.rs b/crates/cellule-runtime/src/cell/actor/state.rs index 07c06269..c0d1e5ca 100644 --- a/crates/cellule-runtime/src/cell/actor/state.rs +++ b/crates/cellule-runtime/src/cell/actor/state.rs @@ -283,6 +283,9 @@ pub(super) struct ActiveCell { pub(super) publisher: Option, pub(super) durability_submitter: CellDurabilitySubmitter, pub(super) publications: VecDeque, + // One observer per resident Cell, retaining only its original receipt. + // Dropping/fencing the Cell cancels observation, never native publication. + pub(super) selection_waiter: Option, pub(super) root_debt: Option, pub(super) materializing: bool, pub(super) publishing_since: Option, @@ -542,6 +545,11 @@ pub(super) enum TaskResult { publisher: Box, result: crate::Result>, }, + BundleSelectionReady { + cell: CellId, + generation: u64, + result: crate::Result<()>, + }, BundleSelected { cell: CellId, generation: u64, diff --git a/crates/cellule-runtime/src/cell/actor/tasks/activation.rs b/crates/cellule-runtime/src/cell/actor/tasks/activation.rs index e74ffd98..55575afe 100644 --- a/crates/cellule-runtime/src/cell/actor/tasks/activation.rs +++ b/crates/cellule-runtime/src/cell/actor/tasks/activation.rs @@ -129,6 +129,7 @@ pub(super) fn handle_activated( publisher: Some(*publisher), durability_submitter, publications: VecDeque::new(), + selection_waiter: None, publishing_since: None, root_debt: None, materializing: false, diff --git a/crates/cellule-runtime/src/cell/actor/tasks/mod.rs b/crates/cellule-runtime/src/cell/actor/tasks/mod.rs index 672130f1..75ceb27b 100644 --- a/crates/cellule-runtime/src/cell/actor/tasks/mod.rs +++ b/crates/cellule-runtime/src/cell/actor/tasks/mod.rs @@ -131,6 +131,11 @@ pub(super) fn handle_task( } => publication::handle_publication_admitted( context, cell, generation, effect_id, publisher, result, ), + TaskResult::BundleSelectionReady { + cell, + generation, + result, + } => publication::handle_selection_ready(context, cell, generation, result), TaskResult::BundleSelected { cell, generation, diff --git a/crates/cellule-runtime/src/cell/actor/tasks/publication.rs b/crates/cellule-runtime/src/cell/actor/tasks/publication.rs index db513e0b..ed7e81cf 100644 --- a/crates/cellule-runtime/src/cell/actor/tasks/publication.rs +++ b/crates/cellule-runtime/src/cell/actor/tasks/publication.rs @@ -2,6 +2,36 @@ use super::*; +pub(super) fn handle_selection_ready( + context: TaskContext<'_>, + cell: CellId, + generation: u64, + result: crate::Result<()>, +) { + let TaskContext { + pool, + cells, + transitioning, + tasks, + node_lease, + .. + } = context; + let Some(active) = cells.get_mut(&cell) else { + return; + }; + if active.generation != generation || active.selection_waiter.take().is_none() { + return; + } + let result = result.and_then(|()| node_lease.check()); + if let Err(error) = result { + tracing::warn!(cell = ?cell, error = ?error, "original selection observation failed"); + fence_active(active); + } else if !active.coordination.is_fenced() { + start_publication(cell, active, pool, tasks); + } + continue_cell(cell, pool, cells, transitioning, tasks, node_lease); +} + pub(super) fn handle_bundle_selected( context: TaskContext<'_>, cell: CellId, diff --git a/crates/cellule-runtime/src/node/bundle/tests/managed.rs b/crates/cellule-runtime/src/node/bundle/tests/managed.rs index e4167e23..e4d91a93 100644 --- a/crates/cellule-runtime/src/node/bundle/tests/managed.rs +++ b/crates/cellule-runtime/src/node/bundle/tests/managed.rs @@ -576,7 +576,7 @@ async fn managed_actor_case( ); } -async fn renew_actor_lease(authority: &Authority, lease: &NodeLeaseGuard) { +pub(super) async fn renew_actor_lease(authority: &Authority, lease: &NodeLeaseGuard) { // Advance the original guard only after a signed authoritative heartbeat. let mut node = authority.observed.lock().await; let mut next = node.advertisement().clone(); diff --git a/crates/cellule-runtime/src/node/bundle/tests/mod.rs b/crates/cellule-runtime/src/node/bundle/tests/mod.rs index a262596a..d4564d79 100644 --- a/crates/cellule-runtime/src/node/bundle/tests/mod.rs +++ b/crates/cellule-runtime/src/node/bundle/tests/mod.rs @@ -22,6 +22,7 @@ mod index; mod lifecycle; mod managed; mod ranges; +mod readiness; mod receipts; mod recovery; struct Fixture { diff --git a/crates/cellule-runtime/src/node/bundle/tests/readiness.rs b/crates/cellule-runtime/src/node/bundle/tests/readiness.rs new file mode 100644 index 00000000..3c9dd5b3 --- /dev/null +++ b/crates/cellule-runtime/src/node/bundle/tests/readiness.rs @@ -0,0 +1,299 @@ +//! A valid Fleet ACK remains usable while origin selection is delayed. +use super::actor::{Authority, transport}; +use super::*; +use crate::cell::actor::CellRuntime; +use crate::cell::catalog::{CatalogEntry, CatalogRole, CellCatalog}; +use crate::cell::executor::{HandlerOutcome, MutationIdentity}; +use crate::cell::worker::SqlWorkerPool; +use crate::fleet::telemetry::{CellTelemetry, CommandResponseSource}; +use crate::identity::{CellTarget, NamespaceId, RequestId, TenantId}; +use crate::node::durability::{ + BundleCheckpoint, NodeBundleAuthority, NodeBundlePublicationAuthority, NodeDurability, +}; +use crate::node::log_shipper::{AssignedCapture, NodeLogShipper}; +use futures_util::future::BoxFuture; +use std::sync::atomic::{AtomicBool, AtomicUsize, Ordering}; +use std::time::Duration; + +struct DelayedSelection { + authority: Arc, + held: AtomicBool, + entered: tokio::sync::Notify, + changed: tokio::sync::Notify, +} + +impl NodeBundleAuthority for DelayedSelection { + fn bind<'a>( + &'a self, + authority: &'a CellAuthority, + observed: &'a VersionedControl, + ) -> BoxFuture<'a, Result> { + NodeBundleAuthority::bind(self.authority.as_ref(), authority, observed) + } + fn close<'a>( + &'a self, + authority: &'a CellAuthority, + observed: &'a VersionedControl, + issued: crate::node::log::CellIssuedRange, + ) -> BoxFuture<'a, Result<()>> { + NodeBundleAuthority::close(self.authority.as_ref(), authority, observed, issued) + } +} + +impl NodeBundlePublicationAuthority for DelayedSelection { + fn select<'a>( + &'a self, + captures: &'a [AssignedCapture], + lease: &'a NodeLeaseGuard, + ) -> BoxFuture<'a, Result>> { + Box::pin(async move { + while self.held.load(Ordering::Acquire) { + let changed = self.changed.notified(); + tokio::pin!(changed); + changed.as_mut().enable(); + self.entered.notify_one(); + if self.held.load(Ordering::Acquire) { + changed.await; + } + } + self.authority.select(captures, lease).await + }) + } + fn checkpoint<'a>(&'a self, checkpoints: &'a [BundleCheckpoint]) -> BoxFuture<'a, Result<()>> { + self.authority.checkpoint(checkpoints) + } +} + +#[derive(Default)] +struct FleetResponses(AtomicUsize); +impl CellTelemetry for FleetResponses { + fn command_response(&self, source: CommandResponseSource, _: Duration, _: Duration) { + if source == CommandResponseSource::Fleet { + self.0.fetch_add(1, Ordering::Relaxed); + } + } +} + +#[tokio::test(flavor = "multi_thread")] +async fn delayed_selection_keeps_fleet_acks_visible_and_allows_prior_root_before_joined_drain() { + let mut f = Fixture::new().await; + super::coverage::enroll(&mut f).await; + let authority = Arc::new(Authority { + directory: f.directory.clone(), + observed: tokio::sync::Mutex::new(f.node.clone()), + }); + let delayed = Arc::new(DelayedSelection { + authority: authority.clone(), + held: AtomicBool::new(false), + entered: tokio::sync::Notify::new(), + changed: tokio::sync::Notify::new(), + }); + let peers = transport(&f, true); + let shipper = NodeLogShipper::new(f.gate.clone(), peers.clone(), Limits::default()).unwrap(); + let durability = Arc::new(NodeDurability::new( + f.gate.clone(), + shipper, + authority.clone(), + peers, + f.lease.clone(), + )); + let pool = SqlWorkerPool::new(2, 4).unwrap(); + let runtime = CellRuntime::new_with_replica_host_requiring_node_lease( + pool.clone(), + 64 << 20, + SessionId::from_bytes([1; 16]), + cellule_ltx::Host::default(), + ) + .unwrap(); + runtime.install_node_lease(f.lease.clone()).unwrap(); + runtime + .install_node_durability(ApplicationId::from_bytes([9; 16]), durability.clone()) + .unwrap(); + durability + .start_bundle_publication(delayed.clone()) + .unwrap(); + let responses = Arc::new(FleetResponses::default()); + runtime.install_telemetry(responses.clone()).unwrap(); + let target = CellTarget::new( + TenantId::from_bytes([1; 16]), + ApplicationId::from_bytes([9; 16]), + NamespaceId::from_bytes([13; 16]), + b"delayed-selection", + ) + .unwrap(); + let catalog = CellCatalog::new(f.layout.clone(), target.tenant()); + let proof = catalog + .provision( + CatalogEntry::new( + &target, + CatalogRole::Application, + Digest::from_bytes([12; 32]), + 1, + ) + .unwrap(), + ) + .await + .unwrap(); + let cell_authority = CellAuthority::new(f.layout.clone()); + let control = cell_authority + .create_initial( + &proof, + IncarnationId::from_bytes([4; 16]), + Owner { + session: SessionId::from_bytes([1; 16]), + endpoint: "https://bundle.internal:8081".into(), + }, + ) + .await + .unwrap(); + let replica = CellReplica::new( + f.layout.clone(), + *target.cell_id().as_bytes(), + [4; 16], + Limits::default(), + ) + .unwrap(); + let handle = runtime + .bootstrap( + proof, + replica.clone(), + cell_authority.clone(), + control, + f.scratch.path().join("delayed.sqlite"), + |tx| { + tx.execute_batch( + "CREATE TABLE counter(value INTEGER); INSERT INTO counter VALUES(0)", + )?; + Ok(()) + }, + ) + .await + .unwrap(); + let digest = Digest::from_bytes([9; 32]); + let mut results = Vec::new(); + for byte in [1, 2] { + let identity = MutationIdentity { + request_id: RequestId::from_bytes([byte; 16]), + issued_at_ms: 10, + expires_at_ms: 10_000, + }; + let result = tokio::time::timeout( + Duration::from_secs(3), + handle.execute(identity, digest, 20, 1024, 1024, |tx| { + tx.execute("UPDATE counter SET value=value+1", [])?; + Ok(HandlerOutcome::Success(b"ready".to_vec())) + }), + ) + .await + .unwrap() + .unwrap(); + results.push((identity, result)); + if byte == 1 { + // The first exact cut is retired; its admitted root debt remains. + tokio::time::timeout(Duration::from_secs(3), async { + while pool.pending(target.cell_id()).await.unwrap().is_some() { + tokio::task::yield_now().await; + } + }) + .await + .unwrap(); + delayed.held.store(true, Ordering::Release); + } + } + tokio::time::timeout(Duration::from_secs(3), delayed.entered.notified()) + .await + .unwrap(); + assert!( + responses.0.load(Ordering::Relaxed) > 0, + "delayed cut must have a real Fleet ACK" + ); + // Exercise the production ten-second wait failure without extending its + // timeout or the original lease. The producer still owns the complete cut. + tokio::time::sleep(Duration::from_secs(11)).await; + super::managed::renew_actor_lease(&authority, &f.lease).await; + assert_eq!( + handle + .query(1024, 1024, |connection| Ok(connection + .query_row("SELECT value FROM counter", [], |row| row.get::<_, i64>(0),)? + .to_le_bytes() + .to_vec())) + .await + .unwrap(), + 2_i64.to_le_bytes() + ); + for (identity, result) in &results { + assert_eq!( + handle + .execute(*identity, digest, 21, 1024, 1024, |_| panic!( + "Fleet retry must not execute twice" + )) + .await + .unwrap(), + *result + ); + } + let shutdown = runtime.shutdown(); + tokio::pin!(shutdown); + // Draining forces the older selected root. It must prepare even though + // the newer capture awaits selection; checkpoint/drain still join later. + tokio::time::timeout(Duration::from_secs(3), async { + tokio::select! { + result = &mut shutdown => panic!("unselected issued cut was not drained: {result:?}"), + () = async { + loop { + let control = cell_authority.load(target.cell_id()).await.unwrap().unwrap(); + if control.value().ltx_root().unwrap().commit_sequence == 1 { break; } + tokio::task::yield_now().await; + } + } => (), + } + }) + .await + .unwrap(); + delayed.held.store(false, Ordering::Release); + delayed.changed.notify_waiters(); + tokio::time::timeout(Duration::from_secs(10), &mut shutdown) + .await + .unwrap() + .unwrap(); + let control = cell_authority + .load(target.cell_id()) + .await + .unwrap() + .unwrap(); + assert_eq!(control.value().state, ControlState::Idle); + assert!(control.value().bundle_binding.is_none()); + let root = control.value().ltx_root().unwrap(); + assert_eq!(root.commit_sequence, 2); + let path = f.scratch.path().join("cold.sqlite"); + replica + .open_root(&root) + .await + .unwrap() + .restore(&path) + .await + .unwrap(); + let cold = cellule_ltx::rusqlite::Connection::open(path).unwrap(); + assert_eq!( + cold.query_row("SELECT value FROM counter", [], |row| row.get::<_, i64>(0)) + .unwrap(), + 2 + ); + assert_eq!( + cold.query_row( + "SELECT COUNT(*) FROM sys_requests WHERE result=?1", + [b"ready".as_slice()], + |row| row.get::<_, i64>(0) + ) + .unwrap(), + 2 + ); + assert_eq!( + pool.resource_ledger() + .snapshot() + .unwrap() + .used + .retained_bytes(), + 0 + ); +} diff --git a/crates/cellule-runtime/src/node/log_shipper/publication/mod.rs b/crates/cellule-runtime/src/node/log_shipper/publication/mod.rs index 8ae99fd4..c38deb2a 100644 --- a/crates/cellule-runtime/src/node/log_shipper/publication/mod.rs +++ b/crates/cellule-runtime/src/node/log_shipper/publication/mod.rs @@ -17,6 +17,11 @@ pub(crate) struct CaptureSelection { } impl CaptureSelection { + /// A scheduling hint only; consumers still verify the original receipt. + pub(crate) fn is_ready(&self) -> bool { + self.receiver.borrow().is_some() + } + pub(crate) async fn selected(&self) -> Result> { let mut receiver = self.receiver.clone(); loop { diff --git a/crates/cellule-runtime/src/publication/mod.rs b/crates/cellule-runtime/src/publication/mod.rs index 4dcd311b..ff412d0b 100644 --- a/crates/cellule-runtime/src/publication/mod.rs +++ b/crates/cellule-runtime/src/publication/mod.rs @@ -1245,6 +1245,14 @@ impl PendingDurability { self.has_bundle_capture() && self.durability.managed_bundle_publication() } + /// Readiness grants no ACK or authority; retirement verifies exact coverage. + pub(crate) fn selection_ready(&self) -> bool { + self.capture + .as_ref() + .and_then(|capture| capture.submitted.selection.as_ref()) + .is_some_and(|selection| selection.is_ready()) + } + pub(crate) async fn selected_capture_prefix(&self) -> Result> { let Some(mut capture) = self.selected_capture().await? else { return Ok(None); From d852525c10a8824cb1bd652df2d57159204f492c Mon Sep 17 00:00:00 2001 From: forhappy Date: Thu, 8 Oct 2026 19:29:11 -0700 Subject: [PATCH 049/102] fix(runtime): preserve checked gate IDs across Rust versions --- crates/cellule-runtime/src/node/log/mod.rs | 20 +++++++++++++++----- 1 file changed, 15 insertions(+), 5 deletions(-) diff --git a/crates/cellule-runtime/src/node/log/mod.rs b/crates/cellule-runtime/src/node/log/mod.rs index 7478d49b..d25df295 100644 --- a/crates/cellule-runtime/src/node/log/mod.rs +++ b/crates/cellule-runtime/src/node/log/mod.rs @@ -382,11 +382,21 @@ impl DurabilityGate { let follower_through = members.iter().map(|member| (*member, 0)).collect(); // Not a persisted identity: exact assignments cannot be confirmed by a // second in-process gate even when boot, epoch and ticket numbers match. - let instance = NEXT_GATE_INSTANCE - .fetch_update(Ordering::Relaxed, Ordering::Relaxed, |next| { - next.checked_add(1) - }) - .map_err(|_| Error::Node("durability gate instance overflow"))?; + let mut instance = NEXT_GATE_INSTANCE.load(Ordering::Relaxed); + loop { + let next = instance + .checked_add(1) + .ok_or(Error::Node("durability gate instance overflow"))?; + match NEXT_GATE_INSTANCE.compare_exchange_weak( + instance, + next, + Ordering::Relaxed, + Ordering::Relaxed, + ) { + Ok(_) => break, + Err(current) => instance = current, + } + } Ok(Self { inner: Arc::new(Mutex::new(GateState { instance, From 11843f6cfb97ea86fb2647a37478039dbfbe0f54 Mon Sep 17 00:00:00 2001 From: forhappy Date: Thu, 8 Oct 2026 19:31:15 -0700 Subject: [PATCH 050/102] docs(perf): record selection-readiness measurements and merge gates --- .../docs/write-performance-design.md | 9 +- docs/bundle-coverage-implementation.md | 8 +- docs/pr67-selection-readiness-measurement.md | 118 ++++++++++++++++++ docs/write-performance-delivery.md | 9 +- 4 files changed, 140 insertions(+), 4 deletions(-) create mode 100644 docs/pr67-selection-readiness-measurement.md diff --git a/crates/cellule-runtime/docs/write-performance-design.md b/crates/cellule-runtime/docs/write-performance-design.md index 88716b82..c5235a56 100644 --- a/crates/cellule-runtime/docs/write-performance-design.md +++ b/crates/cellule-runtime/docs/write-performance-design.md @@ -99,8 +99,13 @@ cleanup. Its unselected suffix remains in the bounded original queue, allowing older root debt to prepare. The existing ten-second cleanup timeout begins after selection readiness, rather than timing an origin wait after Fleet ACKs. Fencing/removal cancels only observation; original native publication and -checkpoint/drain obligations remain joined. Performance qualification of this -change remains required. +checkpoint/drain obligations remain joined. Its real-actor delayed-selection +regression passes read/retry visibility, older-root progress and joined cold +restore. The [paired measurement](../../../docs/pr67-selection-readiness-measurement.md) +at `6c909a6` completes 184.77 Fleet writes/s versus 195.13 before, with failed +application availability. Bucket completes 248.68 writes/s versus 247.55 with +passing ACK audits; its fixture bypasses the managed producer. No throughput +gain or performance qualification is established. A follow-up preserves exact selected suffixes across a confirmed checkpoint, including receipts selected against intermediate bases while older roots were diff --git a/docs/bundle-coverage-implementation.md b/docs/bundle-coverage-implementation.md index 470d8d9f..243607fd 100644 --- a/docs/bundle-coverage-implementation.md +++ b/docs/bundle-coverage-implementation.md @@ -11,7 +11,13 @@ Fleet writes/s from 525.67 and failed availability/drain. The subsequent [coverage-race measurement](pr67-coverage-race-measurement.md) at `e40ecd6` passes all-ACK warm/cold audit and joined drain, but completes 100.20 Fleet writes/s versus 106.05 before the fix: no demonstrated throughput gain. -The latest [release-build repeat](pr67-release-repeat-measurement.md) of the same +The latest [selection-readiness comparison](pr67-selection-readiness-measurement.md) +at `6c909a6` preserves valid Fleet ACK read/retry visibility while selection +waits and allows older roots to prepare, with joined cold recovery in its +real-actor regression. The application completes 184.77 Fleet and 248.68 Bucket +writes/s versus 195.13 and 247.55 before. Fleet availability still fails; +Bucket audits pass but delivery targets fail. No throughput gain is established. +The earlier [release-build repeat](pr67-release-repeat-measurement.md) of the same `4a8f55c` binary completes 107.95 Fleet and 268.12 Bucket writes/s. Fleet still fails warm ACK availability; Bucket audits pass but delivery targets fail. Diagnostic logs identify shared-selection deadlines that fence Cells and diff --git a/docs/pr67-selection-readiness-measurement.md b/docs/pr67-selection-readiness-measurement.md new file mode 100644 index 00000000..f962d077 --- /dev/null +++ b/docs/pr67-selection-readiness-measurement.md @@ -0,0 +1,118 @@ +# PR 67: selection readiness and paired write measurement + +**PR #67 is not ready to merge. The latest candidate completes 184.77 Fleet +writes/s and 248.68 Bucket writes/s. No throughput improvement or celld parity +is established.** Fleet still fails availability; Bucket passes ACK audits but +misses delivery and latency targets. + +## Change and regression + +The previous selection task owned the Cell publisher while waiting for shared +selection. Its ten-second timeout fenced a Cell after valid Fleet ACKs and +prevented older root debt from preparing. Commit +`6c909a6dec983bc222a0e1c047e878808f677cb2` observes authenticated readiness once +per Cell without owning the publisher. Only the ready oldest prefix enters +exact cleanup; the existing cleanup timeout and original proof checks remain. +Fencing cancels observation, while native publication and drain remain joined. + +The real-actor regression holds the second selection for eleven seconds after +two durable filesystem followers grant its ACK. The old implementation fails +with `Fenced`; the new implementation preserves reads and exact retries, +prepares the older root before releasing the held selection, joins shutdown, +cold-restores both mutations/results and releases retained credits. This fixes +the reproduced ownership failure, not application availability under load. + +## Matched diagnostic + +Six fresh cases run sequentially: before, candidate and celld for each mode. +The before binary is `4a8f55cc99414e6a3907531c774a184842e1de3e`; the candidate +is `6c909a6`. Celld is pinned at +`f2bf648663a610eefde71f3547ad61e9b896b1f0`. Identical fixture, client, auditor, +images, host and runner provenance pass comparison checks. Each case has 1,000 +uniform Cells, 96-byte values, SQL INSERT plus SELECT, a two-hour retry/result +ledger, 128 clients/queue slots, a 30-second warmup and 60-second measured window. +Fleet offers 15K writes/s with two followers; Bucket offers 2K/s. + +The ARM64 Docker VM has **8 CPUs and 8 GiB shared RAM**. Serving containers have +8-CPU/16-GiB ceilings and 4-GiB tmpfs; client/provider ceilings exceed VM resources. +The existing external RustFS adaptation, 64-MiB retention budget, 1-GiB managed +disk budget and gates are unchanged. No build or contributor suite overlaps +timed windows. This does not qualify a dedicated 8-vCPU/16-GiB node, physical +device durability, the laptop KV workload or the 2,000-Cell goal. + +TPS counts successful writes completed inside the window. Successful scheduled +p99 includes trailing successes and excludes errors/dropped offers; request +p99 starts at issuance. Qualification also checks all-attempt latency. + +| Mode / system | Successful writes/s | Successful scheduled p99 ms | Successful request p99 ms | Measured errors | Queue drops | +| --- | ---: | ---: | ---: | ---: | ---: | +| Fleet / before | 195.13 | 1,824.85 | 1,120.30 | 319,678 | 568,614 | +| Fleet / candidate | 184.77 | 1,745.68 | 1,037.21 | 328,243 | 560,671 | +| Fleet / celld | 4,717.33 | 183.65 | 90.23 | 3,583 | 613,376 | +| Bucket / before | 247.55 | 4,785.42 | 4,241.42 | 0 | 104,891 | +| Bucket / candidate | 248.68 | 4,512.72 | 4,006.82 | 0 | 104,823 | +| Bucket / celld | 1,321.83 | 1,046.07 | 561.37 | 0 | 40,437 | + +Candidate Fleet completes 5.3% fewer writes than before; Bucket differs by ++0.46%, and its fixture bypasses the managed producer. One short pair does not +establish an attributable change or repeatable gain. All six points fail +qualification. These overloaded completion counts are not sustainable capacities. +Before Fleet also has 20,579 warmup request errors; the other five have zero. +All six drop warmup offers. No failed case is omitted or silently retried. + +## ACK availability and drain + +| Mode / system | Complete ACK cohort | Warm errors / retry checks | Cold read/retry | Successful drain seconds | +| --- | ---: | --- | --- | ---: | +| Fleet / before | 21,844 | 15,692 / 6,152 | not reached | absent | +| Fleet / candidate | 21,366 | 19,797 / 1,569 | not reached | absent | +| Fleet / celld | 464,403 | 464,399 / 4 | not reached | absent | +| Bucket / before | 26,519 | 0 / 26,519 | pass: all 26,519 | 8.58 | +| Bucket / candidate | 26,423 | 0 / 26,423 | pass: all 26,423 | 25.29 | +| Bucket / celld | 124,185 | 0 / 124,185 | pass: all 124,185 | 4.96 | + +Candidate Fleet's four sampled first audit errors are POST retries to `/orders`, +after the auditor's matched GET check. They suggest retry admission under +pressure needs a regression; they do not establish the cause of every failure. +The pre-SQL command gate refuses publication backlog before durable retry lookup. +Cellule owners exit zero during failed-case cleanup, which does not establish +successful aggregate drain. Celld Fleet is OOM-killed, exit 137. Missing later +provider observations fail gates; unavailable audits do not prove mutation loss. +Passing Bucket cold audits follow graceful drain, not failed-owner recovery. + +Candidate Fleet materializes 11.12 commands/root, with 0.5422 PUT attempts and +13.9287 GET/range attempts per completed write. These canonical storage API +boundary deltas include work for earlier cuts; SDK internal retries are not +separately counted. Steady Bundle ACKs remain zero. Retention rises from 27.88 +to 58.40 MiB of 64 MiB, and oldest debt ends at 45,452 ms. Two samples do not +prove bounded debt. The 215-command and conditional 0.05-PUT targets remain unmet. +Bucket still uses per-Cell roots, at 3.6258 PUT attempts/completed write. + +## Verification and merge gates + +The frozen selection-fix snapshot passes all eleven contributor routes: +1,962 tests passed, zero failed and 38 ignored. Those routes pass again on +Rust 1.97 after compatibility commit `d852525` replaces the deprecated atomic +update with an equivalent checked compare-and-swap loop. Rust 1.99 Clippy also +passes for all workspace targets/features with warnings denied. That subsequent +node-initialization change is outside the timed binaries above and is not +presented as a performance improvement. Independent journal replay +reconciles every offer, attempt, success, error, drop and trailing completion; +complete ACK cohort counts and hashes match. Raw journals, binaries, provider +data, failed snapshots and manifests remain outside Git in +`/Volumes/Workspace/crabbuild-target/cellule-write-perf-8ad1`. +All 9,530 entries in the retained measurement evidence index rehash correctly. +Index SHA-256: +`f8b13a0ca0f188212e4f8894d125fd8c4f3eab8366c3d664ec3558a6b48f29ef`. + +| Required before merge | Measurable completion | +| --- | --- | +| CI and compatibility | All required checks green on the final head; retain Rust 1.97 compatibility. `d852525` addresses the previous head's deprecated `AtomicU64::fetch_update`; final-head CI remains required | +| Availability and joined drain | Reads and known retries remain available under backlog; zero ACK-audit errors; complete accepted/issued range joins before departure | +| Shared publication and bounded debt | Connect Bucket producer, demonstrate materializer progress, bounded retained/native debt and complete publication cost; qualify 215-command density and the conditional PUT target | +| Recovery and collection | Kill owner with prior Fleet ACKs, recover every issued suffix, preserve exact retries, transfer safely and collect only from complete cross-Cell references after quiescence/grace | +| Performance qualification | Three matched repetitions of at least five minutes, sustainable capacity search, zero errors/drops, Fleet 15K/s at scheduled p99 ≤50 ms, Bucket 2K/s at ≤200 ms, read-only/mixed guardrails and the 2,000-Cell / 10K-write / 50K-read standard-node goal | + +The [delivery report](write-performance-delivery.md) and +[runtime design](../crates/cellule-runtime/docs/write-performance-design.md) +retain these requirements. Keep PR #67 in draft. diff --git a/docs/write-performance-delivery.md b/docs/write-performance-delivery.md index 3f41805f..c7719bf4 100644 --- a/docs/write-performance-delivery.md +++ b/docs/write-performance-delivery.md @@ -6,7 +6,14 @@ comparison retains exact retry and cold-state audits. **Celld write parity has not been established.** The [proposal](write-performance-proposal.md) remains the acceptance contract; completing tests or a load run does not pass its gates. -The latest [release-build repeat](pr67-release-repeat-measurement.md) measures +The latest [selection-readiness comparison](pr67-selection-readiness-measurement.md) +measures `6c909a6` at **184.77 Fleet writes/s and 248.68 Bucket writes/s**, versus +195.13 and 247.55 before. Its delayed-selection actor regression passes, but +Fleet still returns 328,243 measured errors and fails 19,797 of 21,366 warm +ACK checks. Bucket passes all 26,423 warm/cold checks but drops 104,823 offers. +No throughput improvement or parity is established; PR #67 remains a draft. + +The earlier [release-build repeat](pr67-release-repeat-measurement.md) measures the unchanged `4a8f55c` binary at **107.95 Fleet writes/s and 268.12 Bucket writes/s**. Fleet returns 499,450 measured errors and fails 16,057 of 16,190 warm ACK checks. Bucket passes all 24,880 warm/cold checks but drops 103,657 From 399e908336490063bd0d12bf0383509215225a7d Mon Sep 17 00:00:00 2001 From: forhappy Date: Thu, 8 Oct 2026 20:10:54 -0700 Subject: [PATCH 051/102] perf(runtime): group fresh historical bundle verification --- .../docs/write-performance-design.md | 16 +- crates/cellule-runtime/src/node/bundle/mod.rs | 2 + .../cellule-runtime/src/node/bundle/origin.rs | 7 +- .../cellule-runtime/src/node/bundle/proof.rs | 155 ++++++++---- .../src/node/bundle/selection.rs | 36 +-- .../src/node/bundle/tests/faults.rs | 9 + .../node/bundle/tests/index/history_cohort.rs | 235 ++++++++++++++++++ .../src/node/bundle/tests/index/mod.rs | 1 + .../src/node/bundle/verification/mod.rs | 233 +++++++++++++++++ .../src/node/bundle/verification/tests.rs | 105 ++++++++ .../src/node/durability/publication/mod.rs | 11 +- docs/bundle-coverage-implementation.md | 8 + 12 files changed, 737 insertions(+), 81 deletions(-) create mode 100644 crates/cellule-runtime/src/node/bundle/tests/index/history_cohort.rs create mode 100644 crates/cellule-runtime/src/node/bundle/verification/mod.rs create mode 100644 crates/cellule-runtime/src/node/bundle/verification/tests.rs diff --git a/crates/cellule-runtime/docs/write-performance-design.md b/crates/cellule-runtime/docs/write-performance-design.md index c5235a56..4d6b69fe 100644 --- a/crates/cellule-runtime/docs/write-performance-design.md +++ b/crates/cellule-runtime/docs/write-performance-design.md @@ -53,7 +53,8 @@ feed can now deliver admitted exact bundle receipts through the actor's existing command, worker and read/retry gate. `NodeDurability::start_bundle_publication` now retains one producer under the installed runtime ledger. It selects complete cohorts of at most 64 captures, 64 frames and 4 MiB, with a 1-ms assembly window -and a 20-MiB working reservation, including one bounded fresh origin read. +and a 23-MiB working reservation, including bounded origin-read scratch and +cohort verification metadata. Startup admission precedes installation of the irreversible feed. Selection and exact root checkpoints use the same original binding/heartbeat authority; 512 checkpoint requests are bounded and their @@ -64,10 +65,15 @@ Bucket-only performance adapter bypasses it. Each selection reads its complete new cohort object once from origin, compares every byte with the proposal, then verifies header, shards, histories and native frames from that operation's read. It retains no cross-operation availability -cache. Historical objects and every Cell base dependency still require origin -verification. Selection drops each checked native frame rather than retaining -reconstruction bodies. The additional 4-MiB buffer is charged before installing -the producer; workload retention and protocol bounds remain unchanged. +cache. Historical native locators are grouped by immutable object and offset for that +selection. Each window reads at most 4 MiB and at most twice the useful union of +requested bytes. Up to eight window reads share 4 MiB of scratch admission; +checked frame facts retain no native bodies. Every frame still passes its exact +digest, scope and Cell-chain checks. Every Cell base dependency still requires +serial origin verification. The producer separately admits 3 MiB of bounded +cohort metadata, raising its working charge from 20 to 23 MiB under the unchanged +workload ledger. Fresh proposal bytes are compared in full before sharing the +already admitted proposal. No observation carries availability across selections. The first end-to-end Fleet diagnostic of this connection failed throughput, availability and drain. It is experimental, not performance qualification. diff --git a/crates/cellule-runtime/src/node/bundle/mod.rs b/crates/cellule-runtime/src/node/bundle/mod.rs index 978c7ec6..b0cf49d2 100644 --- a/crates/cellule-runtime/src/node/bundle/mod.rs +++ b/crates/cellule-runtime/src/node/bundle/mod.rs @@ -46,6 +46,7 @@ mod origin; mod proof; pub(crate) mod recovery; mod selection; +mod verification; pub(crate) use continuation::MaterializedBundlePrefix; #[cfg(test)] use proof::checkpoint_prefix; @@ -58,6 +59,7 @@ pub(crate) mod store; mod tests; pub(crate) const MAX_BUNDLE_BYTES: u64 = 4 << 20; +pub(crate) const COHORT_VERIFICATION_BYTES: u64 = 3 << 20; const MAX_BINDINGS: usize = 4_096; const MAX_LOCATORS: usize = 256; const MAX_INLINE_LOCATORS: usize = 32; diff --git a/crates/cellule-runtime/src/node/bundle/origin.rs b/crates/cellule-runtime/src/node/bundle/origin.rs index 79b87a8a..4298315b 100644 --- a/crates/cellule-runtime/src/node/bundle/origin.rs +++ b/crates/cellule-runtime/src/node/bundle/origin.rs @@ -31,11 +31,14 @@ impl OriginBundle { Ok(Self { session: prepared.catalog.session, head: prepared.head, - body, + // The fresh body was compared byte-for-byte above. Sharing the + // admitted proposal now releases the extra origin-read buffer; + // these bytes grant no availability in any later operation. + body: prepared.body.clone(), }) } - fn range( + pub(super) fn range( &self, session: SessionId, epoch: u64, diff --git a/crates/cellule-runtime/src/node/bundle/proof.rs b/crates/cellule-runtime/src/node/bundle/proof.rs index c19ca8ba..ed130794 100644 --- a/crates/cellule-runtime/src/node/bundle/proof.rs +++ b/crates/cellule-runtime/src/node/bundle/proof.rs @@ -128,19 +128,6 @@ pub(super) async fn verify_binding( Ok(frames) } -pub(super) async fn verify_selected_binding( - layout: &cellule_ltx::CellStorageLayout, - session: SessionId, - epoch: u64, - binding: &Binding, - limits: cellule_ltx::Limits, - origin: &origin::OriginBundle, -) -> Result<()> { - // Selection needs exact locators, not retained native bodies. Keep the same - // verifier as reconstruction while dropping each checked frame promptly. - verify_binding_into(layout, session, epoch, binding, limits, Some(origin), None).await -} - async fn verify_binding_into( layout: &cellule_ltx::CellStorageLayout, session: SessionId, @@ -151,19 +138,7 @@ async fn verify_binding_into( mut frames: Option<&mut Vec>, ) -> Result<()> { verify_base(layout, binding, limits).await?; - let mut position = binding - .control - .ltx_root() - .ok_or(Error::Node("bundle base absent"))? - .position; - let mut commit = binding - .control - .root - .as_ref() - .ok_or(Error::Node("bundle base absent"))? - .commit_sequence; - let mut sequence = 0; - let mut first_commit = commit; + let mut chain = BindingChain::new(binding)?; for locator in &binding.locators { let object = locator .object @@ -174,43 +149,113 @@ async fn verify_binding_into( .ok_or(Error::Node("bundle locator overflow"))?; let bytes = origin::read_range(layout, session, epoch, object, locator.offset..end, origin).await?; - if bytes.len() as u64 != locator.bytes - || *blake3::hash(&bytes).as_bytes() != *locator.frame_digest.as_bytes() - { - return Err(Error::Node("bundle frame digest differs")); + let frame = checked_frame(session, epoch, binding, locator, bytes, limits)?; + chain.accept(FrameStep::from_frame(&frame))?; + if let Some(frames) = &mut frames { + frames.push(frame); + } + } + chain.finish(binding) +} + +/// Checked frame facts retained only within one verification operation. They +/// grant no proof or availability outside that operation and hold no body. +#[derive(Clone, Copy)] +pub(super) struct FrameStep { + sequence: u64, + first_commit: u64, + commit: u64, + min_txid: u64, + pre_checksum: u64, + position: cellule_ltx::Position, +} + +impl FrameStep { + pub(super) fn from_frame(frame: &cellule_ltx::VerifiedNodeFrame) -> Self { + Self { + sequence: frame.scope().node_sequence, + first_commit: frame.first_commit_sequence(), + commit: frame.scope().commit_sequence, + min_txid: frame.segment().min_txid, + pre_checksum: frame.segment().pre_checksum, + position: frame.segment().position(), } - let frame = cellule_ltx::inspect_node_frame(bytes, limits)?; - let scope = frame.scope(); - if scope.leader_session != *session.as_bytes() - || scope.log_epoch != epoch - || scope.application != *binding.application.as_bytes() - || scope.cell != *binding.control.cell.as_bytes() - || scope.incarnation != *binding.control.incarnation.as_bytes() - || scope.cell_epoch != binding.control.epoch - || scope.node_sequence <= sequence - || (scope.commit_sequence == commit && frame.first_commit_sequence() != first_commit) - || (scope.commit_sequence != commit - && commit.checked_add(1) != Some(frame.first_commit_sequence())) - || position.txid.checked_add(1) != Some(frame.segment().min_txid) - || position.checksum != frame.segment().pre_checksum + } +} + +pub(super) struct BindingChain { + position: cellule_ltx::Position, + commit: u64, + sequence: u64, + first_commit: u64, +} + +impl BindingChain { + pub(super) fn new(binding: &Binding) -> Result { + let base = binding + .control + .ltx_root() + .ok_or(Error::Node("bundle base absent"))?; + Ok(Self { + position: base.position, + commit: base.commit_sequence, + sequence: 0, + first_commit: base.commit_sequence, + }) + } + + pub(super) fn accept(&mut self, step: FrameStep) -> Result<()> { + if step.sequence <= self.sequence + || (step.commit == self.commit && step.first_commit != self.first_commit) + || (step.commit != self.commit && self.commit.checked_add(1) != Some(step.first_commit)) + || self.position.txid.checked_add(1) != Some(step.min_txid) + || self.position.checksum != step.pre_checksum { return Err(Error::Node("bundle locator violates exact Cell range")); } - first_commit = frame.first_commit_sequence(); - position = frame.segment().position(); - commit = scope.commit_sequence; - sequence = scope.node_sequence; - if let Some(frames) = &mut frames { - frames.push(frame); + self.position = step.position; + self.commit = step.commit; + self.sequence = step.sequence; + self.first_commit = step.first_commit; + Ok(()) + } + + pub(super) fn finish(self, binding: &Binding) -> Result<()> { + if self.position != binding.selected_position + || self.commit != binding.selected_commit + || (!binding.locators.is_empty() && self.sequence != binding.selected_sequence) + { + return Err(Error::Node("bundle proof endpoint differs")); } + Ok(()) + } +} + +pub(super) fn checked_frame( + session: SessionId, + epoch: u64, + binding: &Binding, + locator: &Locator, + bytes: Bytes, + limits: cellule_ltx::Limits, +) -> Result { + if bytes.len() as u64 != locator.bytes + || *blake3::hash(&bytes).as_bytes() != *locator.frame_digest.as_bytes() + { + return Err(Error::Node("bundle frame digest differs")); } - if position != binding.selected_position - || commit != binding.selected_commit - || (!binding.locators.is_empty() && sequence != binding.selected_sequence) + let frame = cellule_ltx::inspect_node_frame(bytes, limits)?; + let scope = frame.scope(); + if scope.leader_session != *session.as_bytes() + || scope.log_epoch != epoch + || scope.application != *binding.application.as_bytes() + || scope.cell != *binding.control.cell.as_bytes() + || scope.incarnation != *binding.control.incarnation.as_bytes() + || scope.cell_epoch != binding.control.epoch { - return Err(Error::Node("bundle proof endpoint differs")); + return Err(Error::Node("bundle locator violates exact Cell range")); } - Ok(()) + Ok(frame) } pub(super) async fn verify_base( diff --git a/crates/cellule-runtime/src/node/bundle/selection.rs b/crates/cellule-runtime/src/node/bundle/selection.rs index 04c80da5..e273e44e 100644 --- a/crates/cellule-runtime/src/node/bundle/selection.rs +++ b/crates/cellule-runtime/src/node/bundle/selection.rs @@ -241,23 +241,27 @@ impl NodeDirectory { Some(&origin), ) .await?; + let bindings = catalog + .bindings + .into_iter() + .filter(|binding| { + cells.contains(&( + *binding.application.as_bytes(), + *binding.control.cell.as_bytes(), + )) + }) + .collect::>(); + verification::verify_cohort( + &self.layout, + catalog.session, + catalog.epoch, + &bindings, + limits, + &origin, + ) + .await?; let mut proofs = Vec::new(); - for binding in catalog.bindings { - if !cells.contains(&( - *binding.application.as_bytes(), - *binding.control.cell.as_bytes(), - )) { - continue; - } - proof::verify_selected_binding( - &self.layout, - catalog.session, - catalog.epoch, - &binding, - limits, - &origin, - ) - .await?; + for binding in bindings { let pin = binding .control .bundle_binding diff --git a/crates/cellule-runtime/src/node/bundle/tests/faults.rs b/crates/cellule-runtime/src/node/bundle/tests/faults.rs index 66b2d2c2..4233cf1d 100644 --- a/crates/cellule-runtime/src/node/bundle/tests/faults.rs +++ b/crates/cellule-runtime/src/node/bundle/tests/faults.rs @@ -12,6 +12,7 @@ pub(super) struct ReplyFault { pub(super) mode: AtomicU8, pub(super) node_updates: AtomicUsize, pub(super) coverage_puts: AtomicUsize, + pub(super) range_started: AtomicUsize, pub(super) pin_started: tokio::sync::Notify, pub(super) pin_resume: tokio::sync::Notify, pub(super) node_started: tokio::sync::Notify, @@ -117,6 +118,14 @@ impl ObjectStore for ReplyFault { self.inner.put_multipart_opts(path, opts).await } async fn get_opts(&self, path: &Path, opts: GetOptions) -> object_store::Result { + if self.mode.load(Ordering::SeqCst) == 11 + && path.as_ref().ends_with(".cnb") + && opts.range.is_some() + { + self.range_started.fetch_add(1, Ordering::SeqCst); + self.node_started.notify_one(); + self.node_resume.notified().await; + } self.inner.get_opts(path, opts).await } fn delete_stream( diff --git a/crates/cellule-runtime/src/node/bundle/tests/index/history_cohort.rs b/crates/cellule-runtime/src/node/bundle/tests/index/history_cohort.rs new file mode 100644 index 00000000..d124e078 --- /dev/null +++ b/crates/cellule-runtime/src/node/bundle/tests/index/history_cohort.rs @@ -0,0 +1,235 @@ +use super::*; + +#[tokio::test] +async fn selection_groups_fresh_historical_extents_across_sixty_four_cells() { + let mut f = Fixture::new().await; + super::super::coverage::enroll(&mut f).await; + let mut cells = Vec::new(); + for number in 4..(4 + MAX_FRAMES as u8) { + cells.push(f.cell(number).await); + f.heartbeat().await; + } + let mut frames = Vec::new(); + let mut assignments = Vec::new(); + for cell in &mut cells { + let (_, capture, assignment) = f.append(cell, 2); + frames.extend(capture); + assignments.push(assignment); + } + let native_bytes = frames + .iter() + .map(|frame| frame.encoded().len()) + .sum::(); + let now = f.node.advertisement().issued_at_ms(); + let first = f + .directory + .prepare_node_bundle(&f.node, &frames, &assignments, now) + .await + .unwrap(); + let (node, _) = f + .directory + .select_node_bundle(&f.node, &first, &f.lease, Limits::default(), now) + .await + .unwrap(); + f.node = node; + let path = f.layout.node_coverage_bundle_path( + f.node.advertisement().session().as_bytes(), + first.head.epoch, + first.head.digest.as_bytes(), + ); + frames.clear(); + assignments.clear(); + for cell in &mut cells { + let (_, capture, assignment) = f.append(cell, 3); + frames.extend(capture); + assignments.push(assignment); + } + let prepared = f + .directory + .prepare_node_bundle(&f.node, &frames, &assignments, now) + .await + .unwrap(); + f.count.reset(); + let (node, proofs) = f + .directory + .select_node_bundle(&f.node, &prepared, &f.lease, Limits::default(), now) + .await + .unwrap(); + let reads = f + .count + .requests() + .into_iter() + .filter(|request| request.location == path.as_ref()) + .collect::>(); + eprintln!( + "historical cohort: cells={} native_bytes={} reads={}", + cells.len(), + native_bytes, + reads.len() + ); + assert_eq!( + reads.len(), + 1, + "one fresh contiguous historical window, rather than a request per Cell" + ); + assert_eq!(proofs.len(), MAX_FRAMES); + for proof in &proofs { + assert_eq!(proof.commit_sequence(), 3); + assert_eq!(proof.locator_count(), 2); + let restored = proof + .recovery_overlay(&f.layout, Limits::default()) + .await + .unwrap(); + assert_eq!(restored.final_commit_sequence(), 3); + let cell = cells + .iter() + .find(|cell| cell.control.value().bundle_binding == Some(proof.binding())) + .unwrap(); + let recovered = cell + .replica + .prepare_recovered_overlay(&restored, 1) + .await + .unwrap(); + let cold = f.scratch.path().join(format!( + "cohort-cold-{}.sqlite", + cell.control.value().cell.as_bytes()[0] + )); + cell.replica + .open_root(&recovered.root()) + .await + .unwrap() + .restore(&cold) + .await + .unwrap(); + let db = rusqlite::Connection::open(cold).unwrap(); + let outcomes: Vec<(String, String)> = db + .prepare("SELECT request, result FROM outcomes ORDER BY request") + .unwrap() + .query_map([], |row| Ok((row.get(0)?, row.get(1)?))) + .unwrap() + .collect::>() + .unwrap(); + assert_eq!( + outcomes, + vec![ + ("request-2".into(), "result-2".into()), + ("request-3".into(), "result-3".into()), + ("seed".into(), "original".into()) + ] + ); + } + f.node = node; + let (body, _) = f + .layout + .store() + .get_with_etag_bounded(&path, MAX_BUNDLE_BYTES) + .await + .unwrap(); + let mut corrupt = body.to_vec(); + corrupt[proofs[0].binding.locators[0].offset as usize + 8] ^= 1; + f.layout + .store() + .put_overwrite(&path, Bytes::from(corrupt)) + .await + .unwrap(); + let puts = f.count.put_requests(); + assert!(matches!( + f.directory + .select_node_bundle(&f.node, &prepared, &f.lease, Limits::default(), now) + .await, + Err(Error::Node("bundle frame digest differs")) + )); + assert_eq!( + f.count.put_requests(), + puts, + "corrupt historical frames select no new authority" + ); + f.layout.store().put_overwrite(&path, body).await.unwrap(); + // A prior selected proof and a proposal still cannot replace a fresh read + // of its historical dependency in a later operation. + f.count.block_body_reads_for(&path); + let puts = f.count.put_requests(); + assert!( + f.directory + .select_node_bundle(&f.node, &prepared, &f.lease, Limits::default(), now) + .await + .is_err() + ); + assert_eq!(f.count.put_requests(), puts); +} + +#[tokio::test] +async fn distinct_historical_reads_overlap_and_cancel_without_selecting_authority() { + use std::sync::atomic::Ordering; + let faults = Arc::new(super::super::faults::ReplyFault::default()); + let mut f = Fixture::with_store(faults.clone()).await; + super::super::coverage::enroll(&mut f).await; + let mut cells = [f.cell(4).await, f.cell(5).await]; + // Each original Cell frame lives in a different immutable object. + for cell in &mut cells { + let (_, frames, assigned) = f.append(cell, 2); + let proposal = f + .directory + .prepare_node_bundle(&f.node, &frames, &[assigned], NOW) + .await + .unwrap(); + f.node = f + .directory + .select_node_bundle(&f.node, &proposal, &f.lease, Limits::default(), NOW) + .await + .unwrap() + .0; + } + let mut frames = Vec::new(); + let mut assigned = Vec::new(); + for cell in &mut cells { + let (_, capture, range) = f.append(cell, 3); + frames.extend(capture); + assigned.push(range); + } + let proposal = f + .directory + .prepare_node_bundle(&f.node, &frames, &assigned, NOW) + .await + .unwrap(); + faults.mode.store(11, Ordering::SeqCst); + f.count.reset(); + { + let selection = + f.directory + .select_node_bundle(&f.node, &proposal, &f.lease, Limits::default(), NOW); + tokio::pin!(selection); + let overlap = async { + loop { + let started = faults.node_started.notified(); + if faults.range_started.load(Ordering::SeqCst) == 2 { + break; + } + started.await; + } + }; + tokio::select! { + _ = &mut selection => panic!("held historical reads must not select"), + result = tokio::time::timeout(std::time::Duration::from_secs(5), overlap) => result.unwrap(), + } + // Drop the pending original selection while both fresh reads wait. + } + assert_eq!( + f.count.put_requests(), + 0, + "cancellation selects no new authority" + ); + faults.mode.store(0, Ordering::SeqCst); + faults.node_resume.notify_waiters(); + let (_, proofs) = f + .directory + .select_node_bundle(&f.node, &proposal, &f.lease, Limits::default(), NOW) + .await + .unwrap(); + assert_eq!( + proofs.len(), + 2, + "a subsequent operation owns fresh scratch admission" + ); + assert!(proofs.iter().all(|proof| proof.commit_sequence() == 3)); +} diff --git a/crates/cellule-runtime/src/node/bundle/tests/index/mod.rs b/crates/cellule-runtime/src/node/bundle/tests/index/mod.rs index 79f7654d..5826a9c0 100644 --- a/crates/cellule-runtime/src/node/bundle/tests/index/mod.rs +++ b/crates/cellule-runtime/src/node/bundle/tests/index/mod.rs @@ -44,4 +44,5 @@ mod cohort; mod compatibility; mod copy_on_write; mod density; +mod history_cohort; mod inventory; diff --git a/crates/cellule-runtime/src/node/bundle/verification/mod.rs b/crates/cellule-runtime/src/node/bundle/verification/mod.rs new file mode 100644 index 00000000..0ab62962 --- /dev/null +++ b/crates/cellule-runtime/src/node/bundle/verification/mod.rs @@ -0,0 +1,233 @@ +//! Fresh cohort reads; exact frame and Cell-chain validation stays canonical. +use super::*; +use futures_util::{StreamExt, stream}; +use std::ops::Range; +use tokio::sync::Semaphore; + +const READ_CONCURRENCY: usize = 8; + +#[derive(Clone, Copy)] +struct ReadIndex { + binding: u16, + locator: u16, +} + +struct Window { + object: Digest, + range: Range, + useful_bytes: u64, + reads: Vec, +} + +fn locator(bindings: &[Binding], read: ReadIndex) -> &Locator { + // Indices are constructed from these immutable slices inside this module. + &bindings[usize::from(read.binding)].locators[usize::from(read.locator)] +} + +struct Windows<'a> { + bindings: &'a [Binding], + reads: std::iter::Peekable>, +} + +fn windows(bindings: &[Binding]) -> Result> { + if bindings.len() > MAX_FRAMES + || bindings + .iter() + .any(|binding| binding.locators.len() > MAX_LOCATORS) + { + return Err(Error::Capacity("bundle cohort verification metadata")); + } + let count = bindings.iter().map(|binding| binding.locators.len()).sum(); + let mut reads = Vec::with_capacity(count); + for (binding, value) in bindings.iter().enumerate() { + for (index, value) in value.locators.iter().enumerate() { + value + .object + .ok_or(Error::Node("bundle locator is unresolved"))?; + value + .offset + .checked_add(value.bytes) + .ok_or(Error::Node("bundle locator overflow"))?; + if value.bytes == 0 || value.bytes > MAX_BUNDLE_BYTES { + return Err(Error::Capacity("bundle historical window bytes")); + } + reads.push(ReadIndex { + binding: u16::try_from(binding) + .map_err(|_| Error::Capacity("bundle cohort verification metadata"))?, + locator: u16::try_from(index) + .map_err(|_| Error::Capacity("bundle cohort verification metadata"))?, + }); + } + } + reads.sort_unstable_by_key(|read| { + let value = locator(bindings, *read); + (value.object.map(|object| *object.as_bytes()), value.offset) + }); + // Keep only compact sorted indices. Windows are assembled lazily, bounded + // by the reader concurrency rather than all 64 * 256 possible extents. + Ok(Windows { + bindings, + reads: reads.into_iter().peekable(), + }) +} + +impl Iterator for Windows<'_> { + type Item = Result; + fn next(&mut self) -> Option { + self.reads.next().map(|read| self.window(read)) + } +} + +impl Windows<'_> { + fn window(&mut self, read: ReadIndex) -> Result { + let value = locator(self.bindings, read); + let mut window = Window { + object: value + .object + .ok_or(Error::Node("bundle locator is unresolved"))?, + range: value.offset..value.offset + value.bytes, + useful_bytes: value.bytes, + reads: vec![read], + }; + while let Some(read) = self.reads.peek().copied() { + let value = locator(self.bindings, read); + if value.object != Some(window.object) { + break; + } + // Every end was checked before this immutable plan was formed. + let end = value.offset + value.bytes; + let useful = window + .useful_bytes + .checked_add(end.saturating_sub(window.range.end.max(value.offset))) + .ok_or(Error::Node("bundle locator overflow"))?; + let span = end.max(window.range.end) - window.range.start; + // Overlapping extents buy no padding; sparse windows cannot read + // more than twice the useful requested union or exceed scratch. + if span > MAX_BUNDLE_BYTES || span > useful.saturating_mul(2) { + break; + } + self.reads.next(); + window.range.end = end.max(window.range.end); + window.useful_bytes = useful; + window.reads.push(read); + } + Ok(window) + } +} + +pub(super) async fn verify_cohort( + layout: &cellule_ltx::CellStorageLayout, + session: SessionId, + epoch: u64, + bindings: &[Binding], + limits: cellule_ltx::Limits, + origin: &origin::OriginBundle, +) -> Result<()> { + let windows = windows(bindings)?; + // Base traversal retains its original serial memory bound and fresh checks. + for binding in bindings { + proof::verify_base(layout, binding, limits).await?; + } + let mut facts = bindings + .iter() + .map(|binding| vec![None; binding.locators.len()]) + .collect::>(); + let scratch = Semaphore::new(MAX_BUNDLE_BYTES as usize); + let mut reads = stream::iter(windows) + .map(|window| async { + read_window( + layout, session, epoch, bindings, limits, origin, &scratch, window?, + ) + .await + }) + .buffer_unordered(READ_CONCURRENCY); + while let Some(result) = reads.next().await { + for (read, step) in result? { + if facts[usize::from(read.binding)][usize::from(read.locator)] + .replace(step) + .is_some() + { + return Err(Error::Node("bundle cohort repeats locator")); + } + } + } + for (binding, facts) in bindings.iter().zip(facts) { + let mut chain = proof::BindingChain::new(binding)?; + for step in facts { + chain.accept(step.ok_or(Error::Node("bundle cohort omits locator"))?)?; + } + chain.finish(binding)?; + } + Ok(()) +} + +#[allow( + clippy::too_many_arguments, + reason = "one read retains its exact scope and shared scratch admission" +)] +async fn read_window( + layout: &cellule_ltx::CellStorageLayout, + session: SessionId, + epoch: u64, + bindings: &[Binding], + limits: cellule_ltx::Limits, + origin: &origin::OriginBundle, + scratch: &Semaphore, + window: Window, +) -> Result> { + let span = window.range.end - window.range.start; + let current = origin.range(session, epoch, window.object, &window.range)?; + let _permit = if current.is_none() { + Some( + scratch + .acquire_many( + u32::try_from(span) + .map_err(|_| Error::Capacity("bundle historical window bytes"))?, + ) + .await + .map_err(|_| Error::RuntimeClosed)?, + ) + } else { + None + }; + let bytes = match current { + Some(bytes) => bytes, + None => { + origin::read_range( + layout, + session, + epoch, + window.object, + window.range.clone(), + None, + ) + .await? + } + }; + if bytes.len() as u64 != span { + return Err(Error::Node("bundle extent is truncated")); + } + let mut facts = Vec::with_capacity(window.reads.len()); + for read in window.reads { + let value = locator(bindings, read); + let start = usize::try_from(value.offset - window.range.start) + .map_err(|_| Error::Node("bundle extent overflow"))?; + let length = + usize::try_from(value.bytes).map_err(|_| Error::Node("bundle extent overflow"))?; + let frame = proof::checked_frame( + session, + epoch, + &bindings[usize::from(read.binding)], + value, + bytes.slice(start..start + length), + limits, + )?; + facts.push((read, proof::FrameStep::from_frame(&frame))); + } + // No native body escapes this operation; the shared scratch permit covers + // each fresh read until every frame in that window has been inspected. + Ok(facts) +} + +#[cfg(test)] +mod tests; diff --git a/crates/cellule-runtime/src/node/bundle/verification/tests.rs b/crates/cellule-runtime/src/node/bundle/verification/tests.rs new file mode 100644 index 00000000..95d94648 --- /dev/null +++ b/crates/cellule-runtime/src/node/bundle/verification/tests.rs @@ -0,0 +1,105 @@ +use super::*; +use crate::control::Owner; +use crate::identity::{ApplicationId, CellId, IncarnationId}; + +fn binding(locators: Vec) -> Binding { + Binding { + application: ApplicationId::from_bytes([9; 16]), + first_commit: 1, + control: Control::initial( + CellId::from_bytes([4; 32]), + IncarnationId::from_bytes([14; 16]), + Owner { + session: SessionId::from_bytes([1; 16]), + endpoint: "https://test.internal".into(), + }, + Digest::from_bytes([12; 32]), + 1, + ) + .unwrap(), + phase: BindingPhase::Open, + terminal: None, + selected_sequence: 1, + selected_commit: 1, + selected_position: cellule_ltx::Position { + txid: 1, + checksum: 2, + }, + locators, + } +} + +fn extent(object: u8, offset: u64, bytes: u64) -> Locator { + Locator { + object: Some(Digest::from_bytes([object; 32])), + offset, + bytes, + frame_digest: Digest::from_bytes([7; 32]), + } +} + +#[test] +fn sparse_ranges_bound_padding_use_union_bytes_and_preserve_every_locator() { + let bindings = [binding(vec![ + extent(1, 0, 10), + extent(1, 5, 10), + extent(1, 25, 10), + extent(1, 500, 10), + extent(2, 0, 10), + ])]; + let planned = windows(&bindings) + .unwrap() + .collect::>>() + .unwrap(); + assert_eq!(planned.len(), 3); + assert_eq!(planned[0].range, 0..35); + assert_eq!( + planned[0].useful_bytes, 25, + "overlap cannot buy extra padding" + ); + assert_eq!(planned[1].range, 500..510); + assert_eq!(planned[2].object, Digest::from_bytes([2; 32])); + assert_eq!( + planned + .iter() + .map(|window| window.reads.len()) + .sum::(), + 5 + ); + for window in &planned { + assert!(window.range.end - window.range.start <= 2 * window.useful_bytes); + } + let large = [binding(vec![ + extent(1, 0, MAX_BUNDLE_BYTES), + extent(1, MAX_BUNDLE_BYTES, 1), + ])]; + assert_eq!( + windows(&large).unwrap().count(), + 2, + "a window never exceeds shared scratch" + ); +} + +#[test] +fn protocol_maximum_verification_payload_fits_its_admitted_metadata_charge() { + // Bound eight lazy windows, original compact indices, doubled growing + // window-read capacities, all frame facts and all returned facts. + // Leave allocator/task overhead inside the remaining admitted headroom. + let count = MAX_FRAMES * MAX_LOCATORS; + let windows = READ_CONCURRENCY * std::mem::size_of::(); + let indices = 3 * count * std::mem::size_of::(); + let facts = count * std::mem::size_of::>(); + let returned = count * std::mem::size_of::<(ReadIndex, proof::FrameStep)>(); + let total = windows + indices + facts + returned; + assert!( + total + 256 * 1024 <= COHORT_VERIFICATION_BYTES as usize, + "payload {total} leaves less than 256 KiB overhead" + ); + let invalid = [binding(vec![extent(1, 0, 0)])]; + assert!(matches!(super::windows(&invalid), Err(Error::Capacity(_)))); + let invalid = [binding(vec![extent(1, u64::MAX, 1)])]; + assert!(matches!( + super::windows(&invalid), + Err(Error::Node("bundle locator overflow")) + )); +} diff --git a/crates/cellule-runtime/src/node/durability/publication/mod.rs b/crates/cellule-runtime/src/node/durability/publication/mod.rs index 3a397885..215362dd 100644 --- a/crates/cellule-runtime/src/node/durability/publication/mod.rs +++ b/crates/cellule-runtime/src/node/durability/publication/mod.rs @@ -81,9 +81,14 @@ impl Publisher { .ok_or(Error::PendingPublication)?; resources.try_reserve( crate::fleet::resource::ResourceCost::zero() - // The fifth bounded buffer is the fresh cohort origin read; - // its bytes are shared only within one verification operation. - .with_retained_bytes((5 * crate::node::bundle::MAX_BUNDLE_BYTES) as usize), + // Fresh origin bytes share the proposal after exact comparison. + // One scratch buffer serves bounded historical windows; checked + // cohort facts/jobs have their own pre-admitted metadata bound. + .with_retained_bytes( + (5 * crate::node::bundle::MAX_BUNDLE_BYTES + + crate::node::bundle::COHORT_VERIFICATION_BYTES) + as usize, + ), ) } diff --git a/docs/bundle-coverage-implementation.md b/docs/bundle-coverage-implementation.md index 243607fd..5d46c168 100644 --- a/docs/bundle-coverage-implementation.md +++ b/docs/bundle-coverage-implementation.md @@ -179,6 +179,14 @@ history, then its exact native frames. A shared shard keeps unrelated histories as authenticated references, including when those bodies are unavailable. Selection verifies every participating Cell's origin and native suffix before CAS; point selection grants no sibling drain or collection authority. +Historical native extents in one selection now share bounded object/offset +windows. A window is at most 4 MiB and twice the useful requested union bytes; +at most eight reads share 4 MiB of scratch. The canonical frame verifier checks +every exact digest and scope, then folds each Cell chain in locator order. +Checked facts retain no bodies or cross-operation availability. Base dependencies +keep their serial fresh verification. The managed producer admits another 3 MiB +for cohort metadata, raising its working reservation from 20 to 23 MiB under the +unchanged workload budget. This request reduction still needs paired TPS evidence. Complete maintenance inventory still verifies every shard, retaining one bounded shard at a time and at most 4,096 duplicate-pin entries. An unresolved history remains a drain obligation. From 22573840a793ef572e0a0bbc47a9c34ae02e823f Mon Sep 17 00:00:00 2001 From: forhappy Date: Thu, 8 Oct 2026 20:38:12 -0700 Subject: [PATCH 052/102] Revert "perf(runtime): group fresh historical bundle verification" This reverts commit 399e908336490063bd0d12bf0383509215225a7d. --- .../docs/write-performance-design.md | 16 +- crates/cellule-runtime/src/node/bundle/mod.rs | 2 - .../cellule-runtime/src/node/bundle/origin.rs | 7 +- .../cellule-runtime/src/node/bundle/proof.rs | 155 ++++-------- .../src/node/bundle/selection.rs | 36 ++- .../src/node/bundle/tests/faults.rs | 9 - .../node/bundle/tests/index/history_cohort.rs | 235 ------------------ .../src/node/bundle/tests/index/mod.rs | 1 - .../src/node/bundle/verification/mod.rs | 233 ----------------- .../src/node/bundle/verification/tests.rs | 105 -------- .../src/node/durability/publication/mod.rs | 11 +- docs/bundle-coverage-implementation.md | 8 - 12 files changed, 81 insertions(+), 737 deletions(-) delete mode 100644 crates/cellule-runtime/src/node/bundle/tests/index/history_cohort.rs delete mode 100644 crates/cellule-runtime/src/node/bundle/verification/mod.rs delete mode 100644 crates/cellule-runtime/src/node/bundle/verification/tests.rs diff --git a/crates/cellule-runtime/docs/write-performance-design.md b/crates/cellule-runtime/docs/write-performance-design.md index 4d6b69fe..c5235a56 100644 --- a/crates/cellule-runtime/docs/write-performance-design.md +++ b/crates/cellule-runtime/docs/write-performance-design.md @@ -53,8 +53,7 @@ feed can now deliver admitted exact bundle receipts through the actor's existing command, worker and read/retry gate. `NodeDurability::start_bundle_publication` now retains one producer under the installed runtime ledger. It selects complete cohorts of at most 64 captures, 64 frames and 4 MiB, with a 1-ms assembly window -and a 23-MiB working reservation, including bounded origin-read scratch and -cohort verification metadata. +and a 20-MiB working reservation, including one bounded fresh origin read. Startup admission precedes installation of the irreversible feed. Selection and exact root checkpoints use the same original binding/heartbeat authority; 512 checkpoint requests are bounded and their @@ -65,15 +64,10 @@ Bucket-only performance adapter bypasses it. Each selection reads its complete new cohort object once from origin, compares every byte with the proposal, then verifies header, shards, histories and native frames from that operation's read. It retains no cross-operation availability -cache. Historical native locators are grouped by immutable object and offset for that -selection. Each window reads at most 4 MiB and at most twice the useful union of -requested bytes. Up to eight window reads share 4 MiB of scratch admission; -checked frame facts retain no native bodies. Every frame still passes its exact -digest, scope and Cell-chain checks. Every Cell base dependency still requires -serial origin verification. The producer separately admits 3 MiB of bounded -cohort metadata, raising its working charge from 20 to 23 MiB under the unchanged -workload ledger. Fresh proposal bytes are compared in full before sharing the -already admitted proposal. No observation carries availability across selections. +cache. Historical objects and every Cell base dependency still require origin +verification. Selection drops each checked native frame rather than retaining +reconstruction bodies. The additional 4-MiB buffer is charged before installing +the producer; workload retention and protocol bounds remain unchanged. The first end-to-end Fleet diagnostic of this connection failed throughput, availability and drain. It is experimental, not performance qualification. diff --git a/crates/cellule-runtime/src/node/bundle/mod.rs b/crates/cellule-runtime/src/node/bundle/mod.rs index b0cf49d2..978c7ec6 100644 --- a/crates/cellule-runtime/src/node/bundle/mod.rs +++ b/crates/cellule-runtime/src/node/bundle/mod.rs @@ -46,7 +46,6 @@ mod origin; mod proof; pub(crate) mod recovery; mod selection; -mod verification; pub(crate) use continuation::MaterializedBundlePrefix; #[cfg(test)] use proof::checkpoint_prefix; @@ -59,7 +58,6 @@ pub(crate) mod store; mod tests; pub(crate) const MAX_BUNDLE_BYTES: u64 = 4 << 20; -pub(crate) const COHORT_VERIFICATION_BYTES: u64 = 3 << 20; const MAX_BINDINGS: usize = 4_096; const MAX_LOCATORS: usize = 256; const MAX_INLINE_LOCATORS: usize = 32; diff --git a/crates/cellule-runtime/src/node/bundle/origin.rs b/crates/cellule-runtime/src/node/bundle/origin.rs index 4298315b..79b87a8a 100644 --- a/crates/cellule-runtime/src/node/bundle/origin.rs +++ b/crates/cellule-runtime/src/node/bundle/origin.rs @@ -31,14 +31,11 @@ impl OriginBundle { Ok(Self { session: prepared.catalog.session, head: prepared.head, - // The fresh body was compared byte-for-byte above. Sharing the - // admitted proposal now releases the extra origin-read buffer; - // these bytes grant no availability in any later operation. - body: prepared.body.clone(), + body, }) } - pub(super) fn range( + fn range( &self, session: SessionId, epoch: u64, diff --git a/crates/cellule-runtime/src/node/bundle/proof.rs b/crates/cellule-runtime/src/node/bundle/proof.rs index ed130794..c19ca8ba 100644 --- a/crates/cellule-runtime/src/node/bundle/proof.rs +++ b/crates/cellule-runtime/src/node/bundle/proof.rs @@ -128,6 +128,19 @@ pub(super) async fn verify_binding( Ok(frames) } +pub(super) async fn verify_selected_binding( + layout: &cellule_ltx::CellStorageLayout, + session: SessionId, + epoch: u64, + binding: &Binding, + limits: cellule_ltx::Limits, + origin: &origin::OriginBundle, +) -> Result<()> { + // Selection needs exact locators, not retained native bodies. Keep the same + // verifier as reconstruction while dropping each checked frame promptly. + verify_binding_into(layout, session, epoch, binding, limits, Some(origin), None).await +} + async fn verify_binding_into( layout: &cellule_ltx::CellStorageLayout, session: SessionId, @@ -138,7 +151,19 @@ async fn verify_binding_into( mut frames: Option<&mut Vec>, ) -> Result<()> { verify_base(layout, binding, limits).await?; - let mut chain = BindingChain::new(binding)?; + let mut position = binding + .control + .ltx_root() + .ok_or(Error::Node("bundle base absent"))? + .position; + let mut commit = binding + .control + .root + .as_ref() + .ok_or(Error::Node("bundle base absent"))? + .commit_sequence; + let mut sequence = 0; + let mut first_commit = commit; for locator in &binding.locators { let object = locator .object @@ -149,113 +174,43 @@ async fn verify_binding_into( .ok_or(Error::Node("bundle locator overflow"))?; let bytes = origin::read_range(layout, session, epoch, object, locator.offset..end, origin).await?; - let frame = checked_frame(session, epoch, binding, locator, bytes, limits)?; - chain.accept(FrameStep::from_frame(&frame))?; - if let Some(frames) = &mut frames { - frames.push(frame); - } - } - chain.finish(binding) -} - -/// Checked frame facts retained only within one verification operation. They -/// grant no proof or availability outside that operation and hold no body. -#[derive(Clone, Copy)] -pub(super) struct FrameStep { - sequence: u64, - first_commit: u64, - commit: u64, - min_txid: u64, - pre_checksum: u64, - position: cellule_ltx::Position, -} - -impl FrameStep { - pub(super) fn from_frame(frame: &cellule_ltx::VerifiedNodeFrame) -> Self { - Self { - sequence: frame.scope().node_sequence, - first_commit: frame.first_commit_sequence(), - commit: frame.scope().commit_sequence, - min_txid: frame.segment().min_txid, - pre_checksum: frame.segment().pre_checksum, - position: frame.segment().position(), + if bytes.len() as u64 != locator.bytes + || *blake3::hash(&bytes).as_bytes() != *locator.frame_digest.as_bytes() + { + return Err(Error::Node("bundle frame digest differs")); } - } -} - -pub(super) struct BindingChain { - position: cellule_ltx::Position, - commit: u64, - sequence: u64, - first_commit: u64, -} - -impl BindingChain { - pub(super) fn new(binding: &Binding) -> Result { - let base = binding - .control - .ltx_root() - .ok_or(Error::Node("bundle base absent"))?; - Ok(Self { - position: base.position, - commit: base.commit_sequence, - sequence: 0, - first_commit: base.commit_sequence, - }) - } - - pub(super) fn accept(&mut self, step: FrameStep) -> Result<()> { - if step.sequence <= self.sequence - || (step.commit == self.commit && step.first_commit != self.first_commit) - || (step.commit != self.commit && self.commit.checked_add(1) != Some(step.first_commit)) - || self.position.txid.checked_add(1) != Some(step.min_txid) - || self.position.checksum != step.pre_checksum + let frame = cellule_ltx::inspect_node_frame(bytes, limits)?; + let scope = frame.scope(); + if scope.leader_session != *session.as_bytes() + || scope.log_epoch != epoch + || scope.application != *binding.application.as_bytes() + || scope.cell != *binding.control.cell.as_bytes() + || scope.incarnation != *binding.control.incarnation.as_bytes() + || scope.cell_epoch != binding.control.epoch + || scope.node_sequence <= sequence + || (scope.commit_sequence == commit && frame.first_commit_sequence() != first_commit) + || (scope.commit_sequence != commit + && commit.checked_add(1) != Some(frame.first_commit_sequence())) + || position.txid.checked_add(1) != Some(frame.segment().min_txid) + || position.checksum != frame.segment().pre_checksum { return Err(Error::Node("bundle locator violates exact Cell range")); } - self.position = step.position; - self.commit = step.commit; - self.sequence = step.sequence; - self.first_commit = step.first_commit; - Ok(()) - } - - pub(super) fn finish(self, binding: &Binding) -> Result<()> { - if self.position != binding.selected_position - || self.commit != binding.selected_commit - || (!binding.locators.is_empty() && self.sequence != binding.selected_sequence) - { - return Err(Error::Node("bundle proof endpoint differs")); + first_commit = frame.first_commit_sequence(); + position = frame.segment().position(); + commit = scope.commit_sequence; + sequence = scope.node_sequence; + if let Some(frames) = &mut frames { + frames.push(frame); } - Ok(()) - } -} - -pub(super) fn checked_frame( - session: SessionId, - epoch: u64, - binding: &Binding, - locator: &Locator, - bytes: Bytes, - limits: cellule_ltx::Limits, -) -> Result { - if bytes.len() as u64 != locator.bytes - || *blake3::hash(&bytes).as_bytes() != *locator.frame_digest.as_bytes() - { - return Err(Error::Node("bundle frame digest differs")); } - let frame = cellule_ltx::inspect_node_frame(bytes, limits)?; - let scope = frame.scope(); - if scope.leader_session != *session.as_bytes() - || scope.log_epoch != epoch - || scope.application != *binding.application.as_bytes() - || scope.cell != *binding.control.cell.as_bytes() - || scope.incarnation != *binding.control.incarnation.as_bytes() - || scope.cell_epoch != binding.control.epoch + if position != binding.selected_position + || commit != binding.selected_commit + || (!binding.locators.is_empty() && sequence != binding.selected_sequence) { - return Err(Error::Node("bundle locator violates exact Cell range")); + return Err(Error::Node("bundle proof endpoint differs")); } - Ok(frame) + Ok(()) } pub(super) async fn verify_base( diff --git a/crates/cellule-runtime/src/node/bundle/selection.rs b/crates/cellule-runtime/src/node/bundle/selection.rs index e273e44e..04c80da5 100644 --- a/crates/cellule-runtime/src/node/bundle/selection.rs +++ b/crates/cellule-runtime/src/node/bundle/selection.rs @@ -241,27 +241,23 @@ impl NodeDirectory { Some(&origin), ) .await?; - let bindings = catalog - .bindings - .into_iter() - .filter(|binding| { - cells.contains(&( - *binding.application.as_bytes(), - *binding.control.cell.as_bytes(), - )) - }) - .collect::>(); - verification::verify_cohort( - &self.layout, - catalog.session, - catalog.epoch, - &bindings, - limits, - &origin, - ) - .await?; let mut proofs = Vec::new(); - for binding in bindings { + for binding in catalog.bindings { + if !cells.contains(&( + *binding.application.as_bytes(), + *binding.control.cell.as_bytes(), + )) { + continue; + } + proof::verify_selected_binding( + &self.layout, + catalog.session, + catalog.epoch, + &binding, + limits, + &origin, + ) + .await?; let pin = binding .control .bundle_binding diff --git a/crates/cellule-runtime/src/node/bundle/tests/faults.rs b/crates/cellule-runtime/src/node/bundle/tests/faults.rs index 4233cf1d..66b2d2c2 100644 --- a/crates/cellule-runtime/src/node/bundle/tests/faults.rs +++ b/crates/cellule-runtime/src/node/bundle/tests/faults.rs @@ -12,7 +12,6 @@ pub(super) struct ReplyFault { pub(super) mode: AtomicU8, pub(super) node_updates: AtomicUsize, pub(super) coverage_puts: AtomicUsize, - pub(super) range_started: AtomicUsize, pub(super) pin_started: tokio::sync::Notify, pub(super) pin_resume: tokio::sync::Notify, pub(super) node_started: tokio::sync::Notify, @@ -118,14 +117,6 @@ impl ObjectStore for ReplyFault { self.inner.put_multipart_opts(path, opts).await } async fn get_opts(&self, path: &Path, opts: GetOptions) -> object_store::Result { - if self.mode.load(Ordering::SeqCst) == 11 - && path.as_ref().ends_with(".cnb") - && opts.range.is_some() - { - self.range_started.fetch_add(1, Ordering::SeqCst); - self.node_started.notify_one(); - self.node_resume.notified().await; - } self.inner.get_opts(path, opts).await } fn delete_stream( diff --git a/crates/cellule-runtime/src/node/bundle/tests/index/history_cohort.rs b/crates/cellule-runtime/src/node/bundle/tests/index/history_cohort.rs deleted file mode 100644 index d124e078..00000000 --- a/crates/cellule-runtime/src/node/bundle/tests/index/history_cohort.rs +++ /dev/null @@ -1,235 +0,0 @@ -use super::*; - -#[tokio::test] -async fn selection_groups_fresh_historical_extents_across_sixty_four_cells() { - let mut f = Fixture::new().await; - super::super::coverage::enroll(&mut f).await; - let mut cells = Vec::new(); - for number in 4..(4 + MAX_FRAMES as u8) { - cells.push(f.cell(number).await); - f.heartbeat().await; - } - let mut frames = Vec::new(); - let mut assignments = Vec::new(); - for cell in &mut cells { - let (_, capture, assignment) = f.append(cell, 2); - frames.extend(capture); - assignments.push(assignment); - } - let native_bytes = frames - .iter() - .map(|frame| frame.encoded().len()) - .sum::(); - let now = f.node.advertisement().issued_at_ms(); - let first = f - .directory - .prepare_node_bundle(&f.node, &frames, &assignments, now) - .await - .unwrap(); - let (node, _) = f - .directory - .select_node_bundle(&f.node, &first, &f.lease, Limits::default(), now) - .await - .unwrap(); - f.node = node; - let path = f.layout.node_coverage_bundle_path( - f.node.advertisement().session().as_bytes(), - first.head.epoch, - first.head.digest.as_bytes(), - ); - frames.clear(); - assignments.clear(); - for cell in &mut cells { - let (_, capture, assignment) = f.append(cell, 3); - frames.extend(capture); - assignments.push(assignment); - } - let prepared = f - .directory - .prepare_node_bundle(&f.node, &frames, &assignments, now) - .await - .unwrap(); - f.count.reset(); - let (node, proofs) = f - .directory - .select_node_bundle(&f.node, &prepared, &f.lease, Limits::default(), now) - .await - .unwrap(); - let reads = f - .count - .requests() - .into_iter() - .filter(|request| request.location == path.as_ref()) - .collect::>(); - eprintln!( - "historical cohort: cells={} native_bytes={} reads={}", - cells.len(), - native_bytes, - reads.len() - ); - assert_eq!( - reads.len(), - 1, - "one fresh contiguous historical window, rather than a request per Cell" - ); - assert_eq!(proofs.len(), MAX_FRAMES); - for proof in &proofs { - assert_eq!(proof.commit_sequence(), 3); - assert_eq!(proof.locator_count(), 2); - let restored = proof - .recovery_overlay(&f.layout, Limits::default()) - .await - .unwrap(); - assert_eq!(restored.final_commit_sequence(), 3); - let cell = cells - .iter() - .find(|cell| cell.control.value().bundle_binding == Some(proof.binding())) - .unwrap(); - let recovered = cell - .replica - .prepare_recovered_overlay(&restored, 1) - .await - .unwrap(); - let cold = f.scratch.path().join(format!( - "cohort-cold-{}.sqlite", - cell.control.value().cell.as_bytes()[0] - )); - cell.replica - .open_root(&recovered.root()) - .await - .unwrap() - .restore(&cold) - .await - .unwrap(); - let db = rusqlite::Connection::open(cold).unwrap(); - let outcomes: Vec<(String, String)> = db - .prepare("SELECT request, result FROM outcomes ORDER BY request") - .unwrap() - .query_map([], |row| Ok((row.get(0)?, row.get(1)?))) - .unwrap() - .collect::>() - .unwrap(); - assert_eq!( - outcomes, - vec![ - ("request-2".into(), "result-2".into()), - ("request-3".into(), "result-3".into()), - ("seed".into(), "original".into()) - ] - ); - } - f.node = node; - let (body, _) = f - .layout - .store() - .get_with_etag_bounded(&path, MAX_BUNDLE_BYTES) - .await - .unwrap(); - let mut corrupt = body.to_vec(); - corrupt[proofs[0].binding.locators[0].offset as usize + 8] ^= 1; - f.layout - .store() - .put_overwrite(&path, Bytes::from(corrupt)) - .await - .unwrap(); - let puts = f.count.put_requests(); - assert!(matches!( - f.directory - .select_node_bundle(&f.node, &prepared, &f.lease, Limits::default(), now) - .await, - Err(Error::Node("bundle frame digest differs")) - )); - assert_eq!( - f.count.put_requests(), - puts, - "corrupt historical frames select no new authority" - ); - f.layout.store().put_overwrite(&path, body).await.unwrap(); - // A prior selected proof and a proposal still cannot replace a fresh read - // of its historical dependency in a later operation. - f.count.block_body_reads_for(&path); - let puts = f.count.put_requests(); - assert!( - f.directory - .select_node_bundle(&f.node, &prepared, &f.lease, Limits::default(), now) - .await - .is_err() - ); - assert_eq!(f.count.put_requests(), puts); -} - -#[tokio::test] -async fn distinct_historical_reads_overlap_and_cancel_without_selecting_authority() { - use std::sync::atomic::Ordering; - let faults = Arc::new(super::super::faults::ReplyFault::default()); - let mut f = Fixture::with_store(faults.clone()).await; - super::super::coverage::enroll(&mut f).await; - let mut cells = [f.cell(4).await, f.cell(5).await]; - // Each original Cell frame lives in a different immutable object. - for cell in &mut cells { - let (_, frames, assigned) = f.append(cell, 2); - let proposal = f - .directory - .prepare_node_bundle(&f.node, &frames, &[assigned], NOW) - .await - .unwrap(); - f.node = f - .directory - .select_node_bundle(&f.node, &proposal, &f.lease, Limits::default(), NOW) - .await - .unwrap() - .0; - } - let mut frames = Vec::new(); - let mut assigned = Vec::new(); - for cell in &mut cells { - let (_, capture, range) = f.append(cell, 3); - frames.extend(capture); - assigned.push(range); - } - let proposal = f - .directory - .prepare_node_bundle(&f.node, &frames, &assigned, NOW) - .await - .unwrap(); - faults.mode.store(11, Ordering::SeqCst); - f.count.reset(); - { - let selection = - f.directory - .select_node_bundle(&f.node, &proposal, &f.lease, Limits::default(), NOW); - tokio::pin!(selection); - let overlap = async { - loop { - let started = faults.node_started.notified(); - if faults.range_started.load(Ordering::SeqCst) == 2 { - break; - } - started.await; - } - }; - tokio::select! { - _ = &mut selection => panic!("held historical reads must not select"), - result = tokio::time::timeout(std::time::Duration::from_secs(5), overlap) => result.unwrap(), - } - // Drop the pending original selection while both fresh reads wait. - } - assert_eq!( - f.count.put_requests(), - 0, - "cancellation selects no new authority" - ); - faults.mode.store(0, Ordering::SeqCst); - faults.node_resume.notify_waiters(); - let (_, proofs) = f - .directory - .select_node_bundle(&f.node, &proposal, &f.lease, Limits::default(), NOW) - .await - .unwrap(); - assert_eq!( - proofs.len(), - 2, - "a subsequent operation owns fresh scratch admission" - ); - assert!(proofs.iter().all(|proof| proof.commit_sequence() == 3)); -} diff --git a/crates/cellule-runtime/src/node/bundle/tests/index/mod.rs b/crates/cellule-runtime/src/node/bundle/tests/index/mod.rs index 5826a9c0..79f7654d 100644 --- a/crates/cellule-runtime/src/node/bundle/tests/index/mod.rs +++ b/crates/cellule-runtime/src/node/bundle/tests/index/mod.rs @@ -44,5 +44,4 @@ mod cohort; mod compatibility; mod copy_on_write; mod density; -mod history_cohort; mod inventory; diff --git a/crates/cellule-runtime/src/node/bundle/verification/mod.rs b/crates/cellule-runtime/src/node/bundle/verification/mod.rs deleted file mode 100644 index 0ab62962..00000000 --- a/crates/cellule-runtime/src/node/bundle/verification/mod.rs +++ /dev/null @@ -1,233 +0,0 @@ -//! Fresh cohort reads; exact frame and Cell-chain validation stays canonical. -use super::*; -use futures_util::{StreamExt, stream}; -use std::ops::Range; -use tokio::sync::Semaphore; - -const READ_CONCURRENCY: usize = 8; - -#[derive(Clone, Copy)] -struct ReadIndex { - binding: u16, - locator: u16, -} - -struct Window { - object: Digest, - range: Range, - useful_bytes: u64, - reads: Vec, -} - -fn locator(bindings: &[Binding], read: ReadIndex) -> &Locator { - // Indices are constructed from these immutable slices inside this module. - &bindings[usize::from(read.binding)].locators[usize::from(read.locator)] -} - -struct Windows<'a> { - bindings: &'a [Binding], - reads: std::iter::Peekable>, -} - -fn windows(bindings: &[Binding]) -> Result> { - if bindings.len() > MAX_FRAMES - || bindings - .iter() - .any(|binding| binding.locators.len() > MAX_LOCATORS) - { - return Err(Error::Capacity("bundle cohort verification metadata")); - } - let count = bindings.iter().map(|binding| binding.locators.len()).sum(); - let mut reads = Vec::with_capacity(count); - for (binding, value) in bindings.iter().enumerate() { - for (index, value) in value.locators.iter().enumerate() { - value - .object - .ok_or(Error::Node("bundle locator is unresolved"))?; - value - .offset - .checked_add(value.bytes) - .ok_or(Error::Node("bundle locator overflow"))?; - if value.bytes == 0 || value.bytes > MAX_BUNDLE_BYTES { - return Err(Error::Capacity("bundle historical window bytes")); - } - reads.push(ReadIndex { - binding: u16::try_from(binding) - .map_err(|_| Error::Capacity("bundle cohort verification metadata"))?, - locator: u16::try_from(index) - .map_err(|_| Error::Capacity("bundle cohort verification metadata"))?, - }); - } - } - reads.sort_unstable_by_key(|read| { - let value = locator(bindings, *read); - (value.object.map(|object| *object.as_bytes()), value.offset) - }); - // Keep only compact sorted indices. Windows are assembled lazily, bounded - // by the reader concurrency rather than all 64 * 256 possible extents. - Ok(Windows { - bindings, - reads: reads.into_iter().peekable(), - }) -} - -impl Iterator for Windows<'_> { - type Item = Result; - fn next(&mut self) -> Option { - self.reads.next().map(|read| self.window(read)) - } -} - -impl Windows<'_> { - fn window(&mut self, read: ReadIndex) -> Result { - let value = locator(self.bindings, read); - let mut window = Window { - object: value - .object - .ok_or(Error::Node("bundle locator is unresolved"))?, - range: value.offset..value.offset + value.bytes, - useful_bytes: value.bytes, - reads: vec![read], - }; - while let Some(read) = self.reads.peek().copied() { - let value = locator(self.bindings, read); - if value.object != Some(window.object) { - break; - } - // Every end was checked before this immutable plan was formed. - let end = value.offset + value.bytes; - let useful = window - .useful_bytes - .checked_add(end.saturating_sub(window.range.end.max(value.offset))) - .ok_or(Error::Node("bundle locator overflow"))?; - let span = end.max(window.range.end) - window.range.start; - // Overlapping extents buy no padding; sparse windows cannot read - // more than twice the useful requested union or exceed scratch. - if span > MAX_BUNDLE_BYTES || span > useful.saturating_mul(2) { - break; - } - self.reads.next(); - window.range.end = end.max(window.range.end); - window.useful_bytes = useful; - window.reads.push(read); - } - Ok(window) - } -} - -pub(super) async fn verify_cohort( - layout: &cellule_ltx::CellStorageLayout, - session: SessionId, - epoch: u64, - bindings: &[Binding], - limits: cellule_ltx::Limits, - origin: &origin::OriginBundle, -) -> Result<()> { - let windows = windows(bindings)?; - // Base traversal retains its original serial memory bound and fresh checks. - for binding in bindings { - proof::verify_base(layout, binding, limits).await?; - } - let mut facts = bindings - .iter() - .map(|binding| vec![None; binding.locators.len()]) - .collect::>(); - let scratch = Semaphore::new(MAX_BUNDLE_BYTES as usize); - let mut reads = stream::iter(windows) - .map(|window| async { - read_window( - layout, session, epoch, bindings, limits, origin, &scratch, window?, - ) - .await - }) - .buffer_unordered(READ_CONCURRENCY); - while let Some(result) = reads.next().await { - for (read, step) in result? { - if facts[usize::from(read.binding)][usize::from(read.locator)] - .replace(step) - .is_some() - { - return Err(Error::Node("bundle cohort repeats locator")); - } - } - } - for (binding, facts) in bindings.iter().zip(facts) { - let mut chain = proof::BindingChain::new(binding)?; - for step in facts { - chain.accept(step.ok_or(Error::Node("bundle cohort omits locator"))?)?; - } - chain.finish(binding)?; - } - Ok(()) -} - -#[allow( - clippy::too_many_arguments, - reason = "one read retains its exact scope and shared scratch admission" -)] -async fn read_window( - layout: &cellule_ltx::CellStorageLayout, - session: SessionId, - epoch: u64, - bindings: &[Binding], - limits: cellule_ltx::Limits, - origin: &origin::OriginBundle, - scratch: &Semaphore, - window: Window, -) -> Result> { - let span = window.range.end - window.range.start; - let current = origin.range(session, epoch, window.object, &window.range)?; - let _permit = if current.is_none() { - Some( - scratch - .acquire_many( - u32::try_from(span) - .map_err(|_| Error::Capacity("bundle historical window bytes"))?, - ) - .await - .map_err(|_| Error::RuntimeClosed)?, - ) - } else { - None - }; - let bytes = match current { - Some(bytes) => bytes, - None => { - origin::read_range( - layout, - session, - epoch, - window.object, - window.range.clone(), - None, - ) - .await? - } - }; - if bytes.len() as u64 != span { - return Err(Error::Node("bundle extent is truncated")); - } - let mut facts = Vec::with_capacity(window.reads.len()); - for read in window.reads { - let value = locator(bindings, read); - let start = usize::try_from(value.offset - window.range.start) - .map_err(|_| Error::Node("bundle extent overflow"))?; - let length = - usize::try_from(value.bytes).map_err(|_| Error::Node("bundle extent overflow"))?; - let frame = proof::checked_frame( - session, - epoch, - &bindings[usize::from(read.binding)], - value, - bytes.slice(start..start + length), - limits, - )?; - facts.push((read, proof::FrameStep::from_frame(&frame))); - } - // No native body escapes this operation; the shared scratch permit covers - // each fresh read until every frame in that window has been inspected. - Ok(facts) -} - -#[cfg(test)] -mod tests; diff --git a/crates/cellule-runtime/src/node/bundle/verification/tests.rs b/crates/cellule-runtime/src/node/bundle/verification/tests.rs deleted file mode 100644 index 95d94648..00000000 --- a/crates/cellule-runtime/src/node/bundle/verification/tests.rs +++ /dev/null @@ -1,105 +0,0 @@ -use super::*; -use crate::control::Owner; -use crate::identity::{ApplicationId, CellId, IncarnationId}; - -fn binding(locators: Vec) -> Binding { - Binding { - application: ApplicationId::from_bytes([9; 16]), - first_commit: 1, - control: Control::initial( - CellId::from_bytes([4; 32]), - IncarnationId::from_bytes([14; 16]), - Owner { - session: SessionId::from_bytes([1; 16]), - endpoint: "https://test.internal".into(), - }, - Digest::from_bytes([12; 32]), - 1, - ) - .unwrap(), - phase: BindingPhase::Open, - terminal: None, - selected_sequence: 1, - selected_commit: 1, - selected_position: cellule_ltx::Position { - txid: 1, - checksum: 2, - }, - locators, - } -} - -fn extent(object: u8, offset: u64, bytes: u64) -> Locator { - Locator { - object: Some(Digest::from_bytes([object; 32])), - offset, - bytes, - frame_digest: Digest::from_bytes([7; 32]), - } -} - -#[test] -fn sparse_ranges_bound_padding_use_union_bytes_and_preserve_every_locator() { - let bindings = [binding(vec![ - extent(1, 0, 10), - extent(1, 5, 10), - extent(1, 25, 10), - extent(1, 500, 10), - extent(2, 0, 10), - ])]; - let planned = windows(&bindings) - .unwrap() - .collect::>>() - .unwrap(); - assert_eq!(planned.len(), 3); - assert_eq!(planned[0].range, 0..35); - assert_eq!( - planned[0].useful_bytes, 25, - "overlap cannot buy extra padding" - ); - assert_eq!(planned[1].range, 500..510); - assert_eq!(planned[2].object, Digest::from_bytes([2; 32])); - assert_eq!( - planned - .iter() - .map(|window| window.reads.len()) - .sum::(), - 5 - ); - for window in &planned { - assert!(window.range.end - window.range.start <= 2 * window.useful_bytes); - } - let large = [binding(vec![ - extent(1, 0, MAX_BUNDLE_BYTES), - extent(1, MAX_BUNDLE_BYTES, 1), - ])]; - assert_eq!( - windows(&large).unwrap().count(), - 2, - "a window never exceeds shared scratch" - ); -} - -#[test] -fn protocol_maximum_verification_payload_fits_its_admitted_metadata_charge() { - // Bound eight lazy windows, original compact indices, doubled growing - // window-read capacities, all frame facts and all returned facts. - // Leave allocator/task overhead inside the remaining admitted headroom. - let count = MAX_FRAMES * MAX_LOCATORS; - let windows = READ_CONCURRENCY * std::mem::size_of::(); - let indices = 3 * count * std::mem::size_of::(); - let facts = count * std::mem::size_of::>(); - let returned = count * std::mem::size_of::<(ReadIndex, proof::FrameStep)>(); - let total = windows + indices + facts + returned; - assert!( - total + 256 * 1024 <= COHORT_VERIFICATION_BYTES as usize, - "payload {total} leaves less than 256 KiB overhead" - ); - let invalid = [binding(vec![extent(1, 0, 0)])]; - assert!(matches!(super::windows(&invalid), Err(Error::Capacity(_)))); - let invalid = [binding(vec![extent(1, u64::MAX, 1)])]; - assert!(matches!( - super::windows(&invalid), - Err(Error::Node("bundle locator overflow")) - )); -} diff --git a/crates/cellule-runtime/src/node/durability/publication/mod.rs b/crates/cellule-runtime/src/node/durability/publication/mod.rs index 215362dd..3a397885 100644 --- a/crates/cellule-runtime/src/node/durability/publication/mod.rs +++ b/crates/cellule-runtime/src/node/durability/publication/mod.rs @@ -81,14 +81,9 @@ impl Publisher { .ok_or(Error::PendingPublication)?; resources.try_reserve( crate::fleet::resource::ResourceCost::zero() - // Fresh origin bytes share the proposal after exact comparison. - // One scratch buffer serves bounded historical windows; checked - // cohort facts/jobs have their own pre-admitted metadata bound. - .with_retained_bytes( - (5 * crate::node::bundle::MAX_BUNDLE_BYTES - + crate::node::bundle::COHORT_VERIFICATION_BYTES) - as usize, - ), + // The fifth bounded buffer is the fresh cohort origin read; + // its bytes are shared only within one verification operation. + .with_retained_bytes((5 * crate::node::bundle::MAX_BUNDLE_BYTES) as usize), ) } diff --git a/docs/bundle-coverage-implementation.md b/docs/bundle-coverage-implementation.md index 5d46c168..243607fd 100644 --- a/docs/bundle-coverage-implementation.md +++ b/docs/bundle-coverage-implementation.md @@ -179,14 +179,6 @@ history, then its exact native frames. A shared shard keeps unrelated histories as authenticated references, including when those bodies are unavailable. Selection verifies every participating Cell's origin and native suffix before CAS; point selection grants no sibling drain or collection authority. -Historical native extents in one selection now share bounded object/offset -windows. A window is at most 4 MiB and twice the useful requested union bytes; -at most eight reads share 4 MiB of scratch. The canonical frame verifier checks -every exact digest and scope, then folds each Cell chain in locator order. -Checked facts retain no bodies or cross-operation availability. Base dependencies -keep their serial fresh verification. The managed producer admits another 3 MiB -for cohort metadata, raising its working reservation from 20 to 23 MiB under the -unchanged workload budget. This request reduction still needs paired TPS evidence. Complete maintenance inventory still verifies every shard, retaining one bounded shard at a time and at most 4,096 duplicate-pin entries. An unresolved history remains a drain obligation. From 433e303a7282a9f83402b70d9e44baa4da764421 Mon Sep 17 00:00:00 2001 From: forhappy Date: Thu, 8 Oct 2026 20:42:40 -0700 Subject: [PATCH 053/102] docs(perf): record withdrawn historical-read experiment --- .../docs/write-performance-design.md | 6 + docs/bundle-coverage-implementation.md | 6 + docs/pr67-historical-read-measurement.md | 126 ++++++++++++++++++ 3 files changed, 138 insertions(+) create mode 100644 docs/pr67-historical-read-measurement.md diff --git a/crates/cellule-runtime/docs/write-performance-design.md b/crates/cellule-runtime/docs/write-performance-design.md index c5235a56..4418e4d7 100644 --- a/crates/cellule-runtime/docs/write-performance-design.md +++ b/crates/cellule-runtime/docs/write-performance-design.md @@ -1,5 +1,11 @@ # Node write and read performance design +The latest [historical-read experiment](../../../docs/pr67-historical-read-measurement.md) +is withdrawn: grouping historical requests passes its isolated regression but +regresses Fleet completion from 201.82 to 18.18 writes/s and fails ACK availability. +Production code returns to `11843f6`. Reuse bounded working credit and reproduce +resource-ledger pressure before accepting another publication optimization. + Status: implementation in progress. The application path is not qualified at the targets below. Component I/O reductions are not application TPS. diff --git a/docs/bundle-coverage-implementation.md b/docs/bundle-coverage-implementation.md index 243607fd..d35ca985 100644 --- a/docs/bundle-coverage-implementation.md +++ b/docs/bundle-coverage-implementation.md @@ -1,5 +1,11 @@ # Bundle coverage implementation +The latest [historical-read experiment](pr67-historical-read-measurement.md) +reduces an isolated 64-Cell historical read cohort from 64 requests to one, but +regresses Fleet completion from 201.82 to 18.18 writes/s and fails ACK availability +with resource-ledger pressure. Commit `399e908` is reverted; production source +returns to baseline `11843f6`. No performance improvement is accepted. + The connected protocol APIs now implement shared selection, independently awaitable root materialization and complete live-writer closure. An explicitly installed original feed can now provide admitted receipts to the actor. diff --git a/docs/pr67-historical-read-measurement.md b/docs/pr67-historical-read-measurement.md new file mode 100644 index 00000000..209db5ff --- /dev/null +++ b/docs/pr67-historical-read-measurement.md @@ -0,0 +1,126 @@ +# PR 67: historical-read experiment and architecture gaps + +**No performance improvement is accepted.** Experimental commit `399e908` groups +fresh historical verification reads, but regresses Fleet completion from 201.82 +to 18.18 writes/s and fails ACK availability. It is reverted by `2257384`. +Production Rust source is byte-identical to the measured baseline `11843f6`. +PR #67 remains a draft; performance parity and merge gates remain unmet. + +## What the experiment established + +The before regression makes 64 historical requests for contiguous native extents +from 64 Cells. The candidate makes one, retaining fresh origin observation, +exact frame digests, scope checks and per-Cell sequence/checksum continuity. +Cold SQLite reconstruction preserves all expected results. Separate tests check +corruption before authority CAS, overlapping reads, cancellation and bounded +verification metadata. These establish request grouping, not application speed. + +The first 24-MiB permanent reservation fails the unchanged 32-MiB managed-runtime +test with publication backlog. A lazy planner reduces it to 23 MiB, versus the +baseline's 20 MiB. That version passes all eleven contributor routes plus +Rust 1.99 Clippy: 1,966 workspace tests and 60 local LTX tests pass, with +38 environment-dependent tests ignored. Initial compile mistakes and the failed +24-MiB test are retained externally. Passing these checks did not predict the +application regression. + +## Matched diagnostic + +Six new sequential cases compare `11843f6cfb97ea86fb2647a37478039dbfbe0f54`, +`399e908336490063bd0d12bf0383509215225a7d`, and celld +`f2bf648663a610eefde71f3547ad61e9b896b1f0`. All use 1,000 uniform Cells, +96-byte SQL values, INSERT plus SELECT, a two-hour outcome/retry ledger, +128 clients/queue slots, 30-second warmup and a 60-second measured window. +Fleet offers 15K writes/s with two followers; Bucket offers 2K/s. +Client/auditor binary, fixture, image, host and runner provenance match. + +The ARM64 Docker VM shares 8 CPUs and 8 GiB RAM across the cluster. Serving +containers have 8-CPU/16-GiB ceilings and 4-GiB tmpfs; ceilings exceed VM resources. +The 64-MiB retained-work and 1-GiB managed-disk budgets remain unchanged. +No build or contributor suite overlaps the timed windows. An initial six-case +setup attempt fails the original one-million-free-inode precheck and produces +no TPS result. Expanding the dedicated VM disk from 120 to 240 GiB retains old +data and restores free inodes. Every measured arm runs after that expansion; +CPU, RAM, workload and gates are unchanged. Both attempts are retained. + +TPS counts successes completed inside the window. Successful scheduled p99 +includes trailing measured successes and excludes errors/drops; request p99 +starts at issuance. Qualification also examines all-attempt latency. + +| Mode / system | Completed writes/s | Successful scheduled p99 ms | Successful request p99 ms | Errors | Queue drops | +| --- | ---: | ---: | ---: | ---: | ---: | +| Fleet / baseline | 201.82 | 1,648.38 | 1,121.67 | 305,973 | 581,854 | +| Fleet / withdrawn candidate | 18.18 | 774.41 | 505.40 | 860,359 | 38,550 | +| Fleet / celld | 4,362.33 | 74.18 | 41.03 | 8,836 | 629,424 | +| Bucket / baseline | 277.25 | 4,254.18 | 3,731.24 | 0 | 103,109 | +| Bucket / withdrawn candidate | 284.88 | 3,665.12 | 3,206.01 | 0 | 102,651 | +| Bucket / celld | 1,288.35 | 1,366.52 | 448.48 | 0 | 42,443 | + +Fleet completion regresses about 91%. Bucket's +2.75% single-pair difference is +not an attributable gain: this adapter bypasses the managed producer being +changed. Every point fails qualification. These overloaded counts do not +establish sustainable capacities or capacity ratios. + +| Mode / system | Complete ACK cohort | Warm errors / retries checked | Cold read/retry | Successful drain seconds | +| --- | ---: | --- | --- | ---: | +| Fleet / baseline | 22,846 | 15,954 / 6,892 | not reached | absent | +| Fleet / withdrawn candidate | 16,032 | 16,032 / 0 | not reached | absent | +| Fleet / celld | 464,790 | 464,790 / 0 | not reached | absent | +| Bucket / baseline | 28,081 | 0 / 28,081 | all pass | 9.30 | +| Bucket / withdrawn candidate | 28,443 | 0 / 28,443 | all pass | 5.36 | +| Bucket / celld | 126,604 | 0 / 126,604 | all pass | 1.16 | + +The candidate's sampled audit failures are GETs. Its owner logs report heartbeat +fencing and drain retrying `Shared(Capacity("resource ledger"))`. The additional +permanent reservation is a headroom hypothesis, not an isolated attribution of +every failure. Celld Fleet is OOM-killed, exit 137. Missing cold audits and joined +drain are unverified gates, not evidence by themselves of mutation loss. + +## Architecture evidence and next action + +The baseline still makes 13.43 GET/range attempts per completed Fleet write, +materializes 12.77 commands per selected root, and ends at 61.15 MiB of the +64-MiB retained-work budget. Oldest publication debt reaches 45,523 ms. +Its mean native capture is 0.186 ms, worker phase 1.180 ms, Fleet proof phase +6.499 ms, Fleet response phase 420.481 ms and publication phase 52,719 ms. +These samples cover different overlapping cohorts and cannot be added as CPU +cost. Bucket materializes about one command/root and makes 3.60 PUT attempts +per completed write. The expensive path is publication and retained work. + +Celld's [bundle and shipping implementation](https://github.com/denoland/celld/blob/f2bf648663a610eefde71f3547ad61e9b896b1f0/crates/celld/ltx_repl.rs) +collects dirty Cell tails into node bundles, credits coverage without immediately +advancing per-Cell materialized roots, and pipelines ordered shipping rounds. +Its [follower stream](https://github.com/denoland/celld/blob/f2bf648663a610eefde71f3547ad61e9b896b1f0/crates/celld/node_log.rs) +groups already delivered appends before the fsync chain. Cellule still awaits +one shipper batch before the next, repeatedly verifies live historical/base +dependencies, and its Bucket adapter publishes through the per-Cell path. +Cellule's managed SQLite sessions already use WAL NORMAL. + +The next experiment must bound verification scratch and metadata within the +existing working credit, pre-admit selected receipt metadata, and preserve +progress headroom through root/checkpoint overlap. Reproduce the resource-ledger +failure before changing it. Pressure must reject new mutations before SQL while +allowing authenticated existing outcomes to be read/retried. Then measure ordered +follower pipelining and Bucket shared selection. None of these proposed changes +has a new qualified TPS result. + +Merge still requires zero-error/drop sustainable capacity, bounded debt, +successful all-ACK warm/cold recovery and joined drain, failed-owner full issued +suffix recovery, safe collection, three paired five-minute repetitions and +read-only/mixed guardrails. The 2,000-Cell / 10K-write / 50K-read target remains +unqualified. + +## Evidence + +Raw journals, binaries, frozen sources, failed attempts and provider observations +remain outside Git under +`/Volumes/Workspace/crabbuild-target/cellule-write-perf-8ad1`. +Canonical reports and independent journal replay reconcile all six measured +cases, every offer/attempt/completion and every complete ACK cohort. The external +`historical-ranges-20261008-evidence-index.json` inventories and hashes the retained +material: 17,318 files / 2,153,161,698 bytes independently rehash. Its SHA-256 +is `65c4706d2c9acdb789d0c17b2e2be5f24f2c51d72e1191db16283f0ce8782b8c`. +The withdrawn implementation remains reproducible by its pinned commit. + +[Previous selection-readiness measurement](pr67-selection-readiness-measurement.md), +[runtime design](../crates/cellule-runtime/docs/write-performance-design.md), +[delivery](write-performance-delivery.md). From addc1cecefb42a094bc6a1c42289d43334c0e645 Mon Sep 17 00:00:00 2001 From: forhappy Date: Thu, 8 Oct 2026 21:07:55 -0700 Subject: [PATCH 054/102] Keep bundle receipt admission progressing through checkpoint pressure --- .../src/node/bundle/tests/mod.rs | 1 + .../src/node/bundle/tests/receipt_pressure.rs | 260 ++++++++++++++++++ .../src/node/durability/mod.rs | 66 +---- .../src/node/durability/publication/mod.rs | 24 +- .../src/node/durability/receipts.rs | 117 ++++++++ docs/pr67-historical-read-measurement.md | 17 +- scripts/perf/README.md | 9 +- 7 files changed, 428 insertions(+), 66 deletions(-) create mode 100644 crates/cellule-runtime/src/node/bundle/tests/receipt_pressure.rs create mode 100644 crates/cellule-runtime/src/node/durability/receipts.rs diff --git a/crates/cellule-runtime/src/node/bundle/tests/mod.rs b/crates/cellule-runtime/src/node/bundle/tests/mod.rs index d4564d79..24248889 100644 --- a/crates/cellule-runtime/src/node/bundle/tests/mod.rs +++ b/crates/cellule-runtime/src/node/bundle/tests/mod.rs @@ -23,6 +23,7 @@ mod lifecycle; mod managed; mod ranges; mod readiness; +mod receipt_pressure; mod receipts; mod recovery; struct Fixture { diff --git a/crates/cellule-runtime/src/node/bundle/tests/receipt_pressure.rs b/crates/cellule-runtime/src/node/bundle/tests/receipt_pressure.rs new file mode 100644 index 00000000..bfac8ace --- /dev/null +++ b/crates/cellule-runtime/src/node/bundle/tests/receipt_pressure.rs @@ -0,0 +1,260 @@ +use super::actor::{Authority, transport}; +use super::*; +use crate::cell::worker::SqlWorkerPool; +use crate::fleet::resource::{ResourceCost, ResourceLedger, ResourceReservation}; +use crate::node::durability::{ + BundleCheckpoint, NodeBundleAuthority, NodeBundlePublicationAuthority, NodeDurability, +}; +use crate::node::log_shipper::{AssignedCapture, NodeLogShipper, NodeLogSubmission}; +use futures_util::future::BoxFuture; +use std::sync::{ + Mutex, + atomic::{AtomicUsize, Ordering}, +}; + +struct HeldReceiptCredit { + original: Arc, + ledger: ResourceLedger, + held: Mutex>, + selects: AtomicUsize, + checkpoints: AtomicUsize, + selected: tokio::sync::Notify, + resume: tokio::sync::Notify, +} + +impl NodeBundleAuthority for HeldReceiptCredit { + fn bind<'a>( + &'a self, + authority: &'a CellAuthority, + observed: &'a VersionedControl, + ) -> BoxFuture<'a, Result> { + NodeBundleAuthority::bind(self.original.as_ref(), authority, observed) + } + fn close<'a>( + &'a self, + authority: &'a CellAuthority, + observed: &'a VersionedControl, + issued: crate::node::log::CellIssuedRange, + ) -> BoxFuture<'a, Result<()>> { + NodeBundleAuthority::close(self.original.as_ref(), authority, observed, issued) + } +} +impl NodeBundlePublicationAuthority for HeldReceiptCredit { + fn select<'a>( + &'a self, + captures: &'a [AssignedCapture], + lease: &'a NodeLeaseGuard, + ) -> BoxFuture<'a, Result>> { + Box::pin(async move { + let proofs = self.original.select(captures, lease).await?; + if self.selects.fetch_add(1, Ordering::SeqCst) == 1 { + let budget = self.ledger.snapshot()?; + let remaining = budget.limit.retained_bytes() - budget.used.retained_bytes(); + // Reproduce an older root admission retaining the remaining + // credit until its canonical checkpoint callback joins. + *self.held.lock().unwrap() = Some( + self.ledger + .try_reserve(ResourceCost::zero().with_retained_bytes(remaining))?, + ); + self.selected.notify_one(); + self.resume.notified().await; + } + Ok(proofs) + }) + } + fn checkpoint<'a>(&'a self, checkpoints: &'a [BundleCheckpoint]) -> BoxFuture<'a, Result<()>> { + Box::pin(async move { + self.original.checkpoint(checkpoints).await?; + self.checkpoints.fetch_add(1, Ordering::SeqCst); + drop(self.held.lock().unwrap().take()); + Ok(()) + }) + } +} + +fn submission(cell: &mut Cell, commit: u64) -> NodeLogSubmission { + cell.db + .transaction(|tx| { + tx.execute( + "INSERT INTO outcomes VALUES(?1,?2)", + [format!("request-{commit}"), format!("result-{commit}")], + ) + }) + .unwrap(); + let cuts = cell.db.capture().unwrap(); + NodeLogSubmission::new( + ApplicationId::from_bytes([9; 16]), + cell.control.value().cell, + cell.control.value().incarnation, + cell.control.value().epoch, + commit, + &cuts, + ) + .unwrap() +} + +#[tokio::test] +async fn producer_waits_for_receipt_credit_and_services_the_checkpoint_that_releases_it() { + let mut f = Fixture::new().await; + super::coverage::enroll(&mut f).await; + let mut cell = f.cell(4).await; + let original = Arc::new(Authority { + directory: f.directory.clone(), + observed: tokio::sync::Mutex::new(f.node.clone()), + }); + let peers = transport(&f, true); + let shipper = NodeLogShipper::new(f.gate.clone(), peers.clone(), Limits::default()).unwrap(); + let durability = Arc::new(NodeDurability::new( + f.gate.clone(), + shipper, + original.clone(), + peers, + f.lease.clone(), + )); + let pool = SqlWorkerPool::new(1, 1).unwrap(); + pool.configure_retained_capacity(32 << 20).unwrap(); + durability + .attach_selection_resources(pool.resource_ledger()) + .unwrap(); + let authority = Arc::new(HeldReceiptCredit { + original, + ledger: pool.resource_ledger(), + held: Mutex::new(None), + selects: AtomicUsize::new(0), + checkpoints: AtomicUsize::new(0), + selected: tokio::sync::Notify::new(), + resume: tokio::sync::Notify::new(), + }); + durability + .start_bundle_publication(authority.clone()) + .unwrap(); + let first = durability + .submit_capture(submission(&mut cell, 2)) + .await + .unwrap(); + let first_selected = first.selection.as_ref().unwrap().selected().await.unwrap(); + let root = f + .publisher(&cell) + .materialize_bundle(&first_selected.proof) + .await + .unwrap(); + let second = durability + .submit_capture(submission(&mut cell, 3)) + .await + .unwrap(); + tokio::time::timeout( + std::time::Duration::from_secs(5), + authority.selected.notified(), + ) + .await + .unwrap(); + assert_eq!( + durability.progress().unwrap().tiered_through, + first.assignment.ticket().last_sequence() + ); + assert_eq!( + pool.resource_ledger() + .snapshot() + .unwrap() + .used + .retained_bytes(), + 32 << 20 + ); + authority.resume.notify_one(); + let checkpoint = + durability.checkpoint_materialized(cell.authority.clone(), root, first_selected.clone()); + let receipt = second.selection.as_ref().unwrap().selected(); + let (checkpoint, receipt) = tokio::time::timeout(std::time::Duration::from_secs(5), async { + tokio::join!(checkpoint, receipt) + }) + .await + .unwrap(); + checkpoint.unwrap(); + let receipt = receipt.unwrap(); + f.lease.check().unwrap(); + assert_eq!(authority.checkpoints.load(Ordering::SeqCst), 1); + assert_eq!( + authority.selects.load(Ordering::SeqCst), + 2, + "credit wait must not repeat the durable selection" + ); + assert_eq!(receipt.proof.commit_sequence(), 3); + assert_eq!( + durability.progress().unwrap().tiered_through, + second.assignment.ticket().last_sequence() + ); + let expected = 20 * 1024 * 1024 + + first_selected.proof.retained_metadata_bytes().unwrap() + + receipt.proof.retained_metadata_bytes().unwrap(); + assert_eq!( + pool.resource_ledger() + .snapshot() + .unwrap() + .used + .retained_bytes(), + expected + ); + cell.control = cell + .authority + .load(cell.control.value().cell) + .await + .unwrap() + .unwrap(); + let root = f + .publisher(&cell) + .materialize_bundle(&receipt.proof) + .await + .unwrap(); + durability + .checkpoint_materialized(cell.authority.clone(), root, receipt.clone()) + .await + .unwrap(); + let cold = f.scratch.path().join("receipt-pressure-cold.sqlite"); + cell.replica + .open_root(&root) + .await + .unwrap() + .restore(&cold) + .await + .unwrap(); + let db = rusqlite::Connection::open(cold).unwrap(); + let outcomes: Vec<(String, String)> = db + .prepare("SELECT request,result FROM outcomes ORDER BY request") + .unwrap() + .query_map([], |row| Ok((row.get(0)?, row.get(1)?))) + .unwrap() + .collect::>() + .unwrap(); + assert_eq!( + outcomes, + vec![ + ("request-2".into(), "result-2".into()), + ("request-3".into(), "result-3".into()), + ("seed".into(), "original".into()) + ] + ); + cell.control = cell + .authority + .load(cell.control.value().cell) + .await + .unwrap() + .unwrap(); + durability + .close_bundle_cell(&cell.authority, &cell.control) + .await + .unwrap(); + durability.shutdown().await.unwrap(); + drop(first); + drop(second); + drop(first_selected); + drop(receipt); + assert_eq!( + pool.resource_ledger() + .snapshot() + .unwrap() + .used + .retained_bytes(), + 0 + ); + pool.shutdown().await.unwrap(); +} diff --git a/crates/cellule-runtime/src/node/durability/mod.rs b/crates/cellule-runtime/src/node/durability/mod.rs index 07c8e0e1..793cb73f 100644 --- a/crates/cellule-runtime/src/node/durability/mod.rs +++ b/crates/cellule-runtime/src/node/durability/mod.rs @@ -19,6 +19,7 @@ mod object_coverage; use object_coverage::ObjectCoverage; mod publication; pub use publication::{BundleCheckpoint, NodeBundlePublicationAuthority}; +mod receipts; /// Original node authority used to enroll and close the Cells of an installed /// shared publication feed. Implementations serialize these mutations and shared @@ -580,68 +581,9 @@ impl NodeDurability { captures: &[crate::node::log_shipper::AssignedCapture], proofs: Vec, ) -> Result { - self.node_lease.check()?; - if captures.is_empty() || captures.len() > 64 || proofs.len() > 64 { - return Err(Error::Capacity("selected capture cohort")); - } - let assignments = proofs - .iter() - .map(|proof| proof.assignment_count()) - .sum::(); - if assignments != captures.len() || proofs.iter().any(|proof| proof.assignment_count() == 0) - { - return Err(Error::Node("selection omits original captured assignments")); - } - for (index, capture) in captures.iter().enumerate() { - let assignment = capture.assignment(); - if captures[..index] - .iter() - .any(|before| before.assignment() == assignment) - || proofs - .iter() - .filter(|proof| proof.contains_assignment(&assignment)) - .count() - != 1 - { - return Err(Error::Node( - "selection differs from original captured cohort", - )); - } - } - let resources = self.selection_resources.get().ok_or(Error::Node( - "bundle publication has no installed runtime resource ledger", - ))?; - let memories = proofs - .iter() - .map(|proof| { - resources.try_reserve( - crate::fleet::resource::ResourceCost::zero() - .with_retained_bytes(proof.retained_metadata_bytes()?), - ) - }) - .collect::>>()?; - // Validate all original gates and leases before sending any receipt. - // Receipt waiters also require the gate's confirmed Bundle source. - let through = - crate::node::bundle::confirm_selected_coverage(&self.gate, &self.node_lease, &proofs)?; - let selected = proofs - .into_iter() - .zip(memories) - .map(|(proof, memory)| { - Arc::new(crate::node::log_shipper::SelectedBundle { - proof, - _memory: memory, - }) - }) - .collect::>(); - for capture in captures { - let proof = selected - .iter() - .find(|selected| selected.proof.contains_assignment(&capture.assignment())) - .ok_or(Error::Node("selected capture lost original assignment"))?; - capture.confirm_selection(Arc::clone(proof)); - } - Ok(crate::node::log_shipper::SelectedBundlePublication { through, selected }) + let cohort = receipts::SelectedCaptures::new(self, captures, proofs)?; + let memory = cohort.resources(self)?.try_reserve(cohort.cost())?; + cohort.confirm(self, memory) } /// Returns this binding's exact enrolled log epoch. diff --git a/crates/cellule-runtime/src/node/durability/publication/mod.rs b/crates/cellule-runtime/src/node/durability/publication/mod.rs index 3a397885..1a2956ad 100644 --- a/crates/cellule-runtime/src/node/durability/publication/mod.rs +++ b/crates/cellule-runtime/src/node/durability/publication/mod.rs @@ -281,7 +281,29 @@ async fn run( proofs = authority.select(&captures, lease) => proofs?, () = lease.wait_fenced() => return Err(Error::Fenced), }; - let selected = original.confirm_selected_captures(&captures, proofs)?; + let cohort = receipts::SelectedCaptures::new(&original, &captures, proofs)?; + let memory = { + let admission = cohort.resources(&original)?.reserve(cohort.cost()); + tokio::pin!(admission); + loop { + tokio::select! { + result = &mut admission => break result?, + checkpoint = checkpoints.recv(), if checkpoints_open => { + if let Some(first) = checkpoint { + // Root tasks retain their admission until this callback + // joins. Servicing it while receipt credit is exhausted + // releases credit without repeating the durable CAS. + tokio::select! { + result = checkpoint_cohort(&authority, &mut checkpoints, first) => result?, + () = lease.wait_fenced() => return Err(Error::Fenced), + } + } else { checkpoints_open = false; } + } + () = lease.wait_fenced() => return Err(Error::Fenced), + } + } + }; + let selected = cohort.confirm(&original, memory)?; progress.send_modify(|state| state.through = selected.selected_through()); drop(selected); drop(captures); diff --git a/crates/cellule-runtime/src/node/durability/receipts.rs b/crates/cellule-runtime/src/node/durability/receipts.rs new file mode 100644 index 00000000..90a6b852 --- /dev/null +++ b/crates/cellule-runtime/src/node/durability/receipts.rs @@ -0,0 +1,117 @@ +//! Complete original capture validation and atomic receipt-memory transfer. +use super::*; +use crate::fleet::resource::{ResourceCost, ResourceLedger, ResourceReservation}; +use crate::node::bundle::BundleCoverageProof; +use crate::node::log_shipper::{AssignedCapture, SelectedBundle, SelectedBundlePublication}; + +pub(super) struct SelectedCaptures<'a> { + captures: &'a [AssignedCapture], + proofs: Vec, + bytes: Vec, + total: usize, +} + +impl<'a> SelectedCaptures<'a> { + pub(super) fn new( + durability: &NodeDurability, + captures: &'a [AssignedCapture], + proofs: Vec, + ) -> Result { + durability.node_lease.check()?; + if captures.is_empty() || captures.len() > 64 || proofs.len() > 64 { + return Err(Error::Capacity("selected capture cohort")); + } + let assignments = proofs + .iter() + .try_fold(0_usize, |sum, proof| { + sum.checked_add(proof.assignment_count()) + }) + .ok_or(Error::Capacity("selected capture assignments"))?; + if assignments != captures.len() || proofs.iter().any(|proof| proof.assignment_count() == 0) + { + return Err(Error::Node("selection omits original captured assignments")); + } + for (index, capture) in captures.iter().enumerate() { + let assignment = capture.assignment(); + if captures[..index] + .iter() + .any(|before| before.assignment() == assignment) + || proofs + .iter() + .filter(|proof| proof.contains_assignment(&assignment)) + .count() + != 1 + { + return Err(Error::Node( + "selection differs from original captured cohort", + )); + } + } + let bytes = proofs + .iter() + .map(BundleCoverageProof::retained_metadata_bytes) + .collect::>>()?; + let total = bytes + .iter() + .try_fold(0_usize, |sum, bytes| sum.checked_add(*bytes)) + .ok_or(Error::Capacity("selected bundle metadata"))?; + Ok(Self { + captures, + proofs, + bytes, + total, + }) + } + + pub(super) fn resources<'b>( + &self, + durability: &'b NodeDurability, + ) -> Result<&'b ResourceLedger> { + durability.selection_resources.get().ok_or(Error::Node( + "bundle publication has no installed runtime resource ledger", + )) + } + + pub(super) fn cost(&self) -> ResourceCost { + ResourceCost::zero().with_retained_bytes(self.total) + } + + pub(super) fn confirm( + self, + durability: &NodeDurability, + mut memory: ResourceReservation, + ) -> Result { + // Transfer a single cohort admission before waking any sibling. Waiting + // for credit may have joined older checkpoints, so revalidate the + // original lease/gate against these exact live proofs after the wait. + let memories = self + .bytes + .into_iter() + .map(|bytes| memory.split_retained(bytes)) + .collect::>>()?; + let through = crate::node::bundle::confirm_selected_coverage( + &durability.gate, + &durability.node_lease, + &self.proofs, + )?; + let selected = self + .proofs + .into_iter() + .zip(memories) + .map(|(proof, memory)| { + Arc::new(SelectedBundle { + proof, + _memory: memory, + }) + }) + .collect::>(); + for capture in self.captures { + let proof = selected + .iter() + .find(|selected| selected.proof.contains_assignment(&capture.assignment())) + .ok_or(Error::Node("selected capture lost original assignment"))?; + capture.confirm_selection(Arc::clone(proof)); + } + Ok(SelectedBundlePublication { through, selected }) + } +} diff --git a/docs/pr67-historical-read-measurement.md b/docs/pr67-historical-read-measurement.md index 209db5ff..0bf9ed8b 100644 --- a/docs/pr67-historical-read-measurement.md +++ b/docs/pr67-historical-read-measurement.md @@ -3,7 +3,7 @@ **No performance improvement is accepted.** Experimental commit `399e908` groups fresh historical verification reads, but regresses Fleet completion from 201.82 to 18.18 writes/s and fails ACK availability. It is reverted by `2257384`. -Production Rust source is byte-identical to the measured baseline `11843f6`. +After that revert, production Rust matched the measured baseline `11843f6`. PR #67 remains a draft; performance parity and merge gates remain unmet. ## What the experiment established @@ -23,7 +23,7 @@ Rust 1.99 Clippy: 1,966 workspace tests and 60 local LTX tests pass, with 24-MiB test are retained externally. Passing these checks did not predict the application regression. -## Matched diagnostic +## Workload-matched diagnostic Six new sequential cases compare `11843f6cfb97ea86fb2647a37478039dbfbe0f54`, `399e908336490063bd0d12bf0383509215225a7d`, and celld @@ -36,6 +36,12 @@ Client/auditor binary, fixture, image, host and runner provenance match. The ARM64 Docker VM shares 8 CPUs and 8 GiB RAM across the cluster. Serving containers have 8-CPU/16-GiB ceilings and 4-GiB tmpfs; ceilings exceed VM resources. The 64-MiB retained-work and 1-GiB managed-disk budgets remain unchanged. +These are explicit Cellule admissions. The runner records their profile values +in every case, but supplies `PARITY_RETAINED_BYTES` and `PARITY_DISK_BYTES` only +to Cellule. It does not configure equivalent celld internal budgets. Equal +case metadata therefore does not establish equal effective resource policies; +the results compare this workload and configuration, not intrinsic maximum +capacity with matched internal memory limits. No build or contributor suite overlaps the timed windows. An initial six-case setup attempt fails the original one-million-free-inode precheck and produces no TPS result. Expanding the dedicated VM disk from 120 to 240 GiB retains old @@ -95,6 +101,13 @@ one shipper batch before the next, repeatedly verifies live historical/base dependencies, and its Bucket adapter publishes through the per-Cell path. Cellule's managed SQLite sessions already use WAL NORMAL. +The measured producer reserves receipt metadata after durable selection. An +isolated real-producer regression with older checkpoint work holding the rest +of a 32-MiB ledger reproduces terminal receipt admission: the first execution +returns `Shared(Capacity("resource ledger"))`, and two repeats observe the +resulting closed checkpoint channel. This establishes a pressure failure at +that seam; it does not attribute every earlier workload failure to this cause. + The next experiment must bound verification scratch and metadata within the existing working credit, pre-admit selected receipt metadata, and preserve progress headroom through root/checkpoint overlap. Reproduce the resource-ledger diff --git a/scripts/perf/README.md b/scripts/perf/README.md index f85b91e2..1f7b818b 100644 --- a/scripts/perf/README.md +++ b/scripts/perf/README.md @@ -6,12 +6,19 @@ and a two-hour durable request/result ledger. It is **SQL application parity**; it does not reproduce a bounded KV upsert benchmark. Use a dedicated Linux Docker context. The current shared-VM profile gives -nodes an 8-CPU/16-GiB ceiling, tmpfs state of 4 GiB, a 2-CPU/2-GiB RustFS +nodes an 8-CPU/16-GiB ceiling, tmpfs state of 4 GiB, a 2-CPU/8-GiB RustFS provider, and a 4-CPU/4-GiB client. These ceilings exceed the shared VM's total CPU; report contention. tmpfs does not qualify physical-device durability. HTTP endpoints, certificates, credentials, and placement are fixture policy. The credentials in these scripts are synthetic and used only by this fixture. +The default 64-MiB retained-work and 1-GiB managed-disk limits are passed only +to Cellule. Case metadata records these profile values for both systems, but +the runner does not configure equivalent celld internal budgets. Matching +workload and container ceilings does not establish matching effective memory +admission or disk policies. Report this asymmetry when comparing results; +neither overloaded completions nor celld OOM establish sustainable capacity. + ## Build and run All artifacts, build caches, binaries, logs and audit journals go From 6dfde17564b4b6863ed5e4e1b1b1ffef0bc474cc Mon Sep 17 00:00:00 2001 From: forhappy Date: Thu, 8 Oct 2026 21:32:38 -0700 Subject: [PATCH 055/102] Record receipt-pressure verification and qualify architecture comparisons --- docs/bundle-coverage-implementation.md | 84 ++++--------- docs/pr67-receipt-admission-measurement.md | 135 +++++++++++++++++++++ 2 files changed, 158 insertions(+), 61 deletions(-) create mode 100644 docs/pr67-receipt-admission-measurement.md diff --git a/docs/bundle-coverage-implementation.md b/docs/bundle-coverage-implementation.md index d35ca985..4bece97b 100644 --- a/docs/bundle-coverage-implementation.md +++ b/docs/bundle-coverage-implementation.md @@ -1,66 +1,20 @@ # Bundle coverage implementation -The latest [historical-read experiment](pr67-historical-read-measurement.md) -reduces an isolated 64-Cell historical read cohort from 64 requests to one, but -regresses Fleet completion from 201.82 to 18.18 writes/s and fails ACK availability -with resource-ledger pressure. Commit `399e908` is reverted; production source -returns to baseline `11843f6`. No performance improvement is accepted. - -The connected protocol APIs now implement shared selection, independently -awaitable root materialization and complete live-writer closure. An explicitly -installed original feed can now provide admitted receipts to the actor. -`NodeDurability::start_bundle_publication` retains a bounded producer and joins -exact checkpoint callbacks under the original authority. The Fleet SQL example -installs it; Bucket-only performance wiring bypasses it. Its first -[measured connection](pr67-managed-producer-measurement.md) regressed to 93.57 -Fleet writes/s from 525.67 and failed availability/drain. The subsequent -[coverage-race measurement](pr67-coverage-race-measurement.md) at `e40ecd6` -passes all-ACK warm/cold audit and joined drain, but completes 100.20 Fleet -writes/s versus 106.05 before the fix: no demonstrated throughput gain. -The latest [selection-readiness comparison](pr67-selection-readiness-measurement.md) -at `6c909a6` preserves valid Fleet ACK read/retry visibility while selection -waits and allows older roots to prepare, with joined cold recovery in its -real-actor regression. The application completes 184.77 Fleet and 248.68 Bucket -writes/s versus 195.13 and 247.55 before. Fleet availability still fails; -Bucket audits pass but delivery targets fail. No throughput gain is established. -The earlier [release-build repeat](pr67-release-repeat-measurement.md) of the same -`4a8f55c` binary completes 107.95 Fleet and 268.12 Bucket writes/s. Fleet still -fails warm ACK availability; Bucket audits pass but delivery targets fail. -Diagnostic logs identify shared-selection deadlines that fence Cells and -publication backlog refusals. No acceptable improvement or parity is established. -The earlier [checkpoint-continuity comparison](pr67-checkpoint-continuity-measurement.md) -at `4a8f55c` verifies live writes across a confirmed root and successive -intermediate-base receipts, with bounded, admitted prefix witnesses. It measures -185.83 Fleet writes/s versus 164.73, but still returns 304,151 errors and fails -warm ACK availability. Bucket completes 215.08 writes/s versus 227.15 with -passing audits and higher p99. All contributor checks pass; no acceptable -performance improvement or parity is established. PR #67 remains a draft. - -The earlier [asynchronous-root comparison](pr67-async-root-measurement.md) at -`6d62d41` retires exact selected captures and schedules admitted roots separately. -It completes 183.65 Fleet writes/s versus 115.27 before, but returns 297,811 -measured request errors and fails its warm ACK audit. Cold recovery and successful -drain are unverified. Root density rises to 10.25, but retained pressure and -publication age grow. This is an availability regression; PR #67 is a draft. -The earlier [cohort-origin comparison](pr67-cohort-origin-measurement.md) at -`9d4e632` reduces one fresh 64-Cell bundle's origin reads from 187 to one. -It completes 95.35 Fleet writes/s versus 88.28 in a fresh paired window, with -passing ACK audits and drain. Total GET/range work remains near 20.4 requests -per completed write, root density is 1.08, and steady Bundle ACKs remain zero. -The producer charges its new buffer under the unchanged retention budget. -Every point fails qualification; a repeatable throughput gain remains unproved. -This slice establishes ordering and reconstruction evidence. -The [earlier application-path benchmark](pr67-performance-reevaluation.md) -measures `7fc0793`; it does not exercise bundle-based responses or establish -write parity. The [WAL NORMAL comparison](pr67-normal-wal-reevaluation.md) -separately records seven completed diagnostic cases and an interrupted matrix. - -The [earlier sparse-root coverage measurement](pr67-sparse-root-coverage-measurement.md) -exercises the ordinary application path at `b185672`: 378.88 completed writes/s -versus 282.43 before the change in one five-minute Fleet-configured pair. -Successful-write p99 improved, median latency worsened, and audit, drain and -debt gates failed. The paired celld owner exited during load. This is diagnostic -evidence, not repeatable improvement or qualified parity. +The [latest receipt-pressure diagnostic](pr67-receipt-admission-measurement.md) +reproduces a fatal post-selection admission failure and verifies waiting for +credit while joining older checkpoints. One fresh Fleet pair completes 154.72 +writes/s versus 122.97 before and 4,001.65 for celld. All three fail warm ACK +availability and qualification; resource policies are asymmetric. No repeatable +throughput improvement or parity is established. PR #67 remains a draft. + +The installed original producer provides admitted shared receipts, independent +root materialization and complete live-writer closure. The Fleet SQL example +installs it; Bucket benchmark wiring bypasses it. Remaining gaps are publication +and verification cost, pressure-safe read/retry and materializer progress, +ordered shipping, safe collection and full lifecycle/performance qualification. +The [historical-read experiment](pr67-historical-read-measurement.md) was reverted +after a severe Fleet regression. Earlier measured slices remain linked from the +[delivery record](write-performance-delivery.md). ## Celld reference and Cellule adaptation @@ -96,6 +50,14 @@ Uploading an immutable object supplies no coverage proof. | Node withdrawal/maintenance | Refuse unresolved bindings; stale fencing preserves the catalog head; one bundle-bound boot cannot rotate its native log to another epoch | | Backup and collection | Backup refuses bound Cells. Coverage objects have no deletion path; this is retention, not a qualified collection implementation | +The producer admits receipt metadata for the complete original cohort together. +When credit is exhausted, it retains that verified cohort and services canonical +checkpoint callbacks whose joining can release older materializer credit. It +then transfers the admitted memory and rechecks the original live gate/lease +before confirming coverage. It never repeats durable selection to obtain credit. +This fixes a reproduced terminal capacity failure; sustained progress and +pressure-safe read/retry still require application qualification. + Live confirmation now reports `DurabilitySource::Bundle` separately from a materialized Cell root. Adjacent exact ranges merge into compact source intervals; root-only gaps keep their original source. A later exact root CAS diff --git a/docs/pr67-receipt-admission-measurement.md b/docs/pr67-receipt-admission-measurement.md new file mode 100644 index 00000000..3964b124 --- /dev/null +++ b/docs/pr67-receipt-admission-measurement.md @@ -0,0 +1,135 @@ +# PR 67: receipt admission and architecture diagnosis + +**Performance parity remains unmet.** Candidate `addc1cecefb42a094bc6a1c42289d43334c0e645` +fixes a reproduced producer pressure failure. A fresh Fleet diagnostic completes +154.72 writes/s versus 122.97 before and 4,001.65 for celld. All three fail +qualification and warm ACK audit. One overloaded pair establishes neither a +repeatable performance improvement nor either system's sustainable capacity. + +## Reproduced failure and change + +The producer selects durable coverage, then reserves receipt metadata. An older +root can retain the remaining credit until its checkpoint callback joins. +Previously, exhausting that credit terminated the producer and fenced the node. +The real producer/checkpoint regression fails three times before the fix: the +first observes `Shared(Capacity("resource ledger"))`; two repeats observe its +closed checkpoint channel. Initial test/build mistakes are retained externally. + +The candidate admits all receipt metadata atomically. When credit is unavailable, +it retains the original selected cohort and services canonical checkpoint +callbacks while waiting. It does not repeat durable selection. It transfers the +admitted credit into shared receipts and rechecks the original live gate and +lease before confirming coverage. The synchronous confirmation API retains its +immediate-capacity contract. The producer's 20-MiB working credit, 32-MiB regression +budget and 64-MiB application budget remain unchanged. + +The regression passes three times after the fix, including a real intermediate +checkpoint, exact later receipt, cold SQLite results and zero retained credit +after joined closure/shutdown. All eleven contributor checks and Rust 1.99 +Clippy pass in the immutable final verification snapshot: 1,963 workspace tests +and 60 local LTX tests pass; 38 environment-dependent tests remain ignored. +Production Rust/Cargo bytes match the measured pinned commit. Documentation +corrections also pass syntax and link checks. + +## Fresh Fleet diagnostic + +Baseline: `11843f6cfb97ea86fb2647a37478039dbfbe0f54`. +Celld: `f2bf648663a610eefde71f3547ad61e9b896b1f0`. +The three sequential cases use identical clients, auditors, fixture bytes and +images: 1,000 uniform Cells, 96-byte SQL values, INSERT plus SELECT and a two-hour +outcome ledger, 128 clients/queue slots, 30-second warmup, 60-second window and +15K offered writes/s. No build or contributor suite overlaps a timed window. + +| System | Completed writes/s | Successful scheduled p99 ms | Successful request p99 ms | Errors | Queue drops | +| --- | ---: | ---: | ---: | ---: | ---: | +| Cellule baseline | 122.97 | 3,087.23 | 1,955.00 | 594,959 | 297,663 | +| Cellule candidate | 154.72 | 2,754.51 | 1,546.26 | 497,981 | 392,480 | +| celld | 4,001.65 | 88.31 | 37.53 | 113,787 | 546,114 | + +TPS counts successes completed inside the window. Successful p99 includes +trailing measured successes and excludes errors/drops; scheduled latency starts +at offered arrival. Qualification also checks all-attempt latency. Candidate +has 9,283 successes inside and 256 after the window. The observed completion +increase is 25.82%; it is not an accepted sustainable throughput gain. + +| System | Complete ACK cohort | Warm errors / completed retries | Cold audit | Qualified drain | +| --- | ---: | ---: | --- | --- | +| Cellule baseline | 16,664 | 3,546 / 13,118 | not reached | unverified | +| Cellule candidate | 20,098 | 5,318 / 14,780 | not reached | unverified | +| celld | 469,993 | 452,930 / 17,063 | not reached | unverified | + +The warm failure fraction worsens from 21.28% to 26.46% in this pair. Higher +completion does not establish an availability improvement. + +Sampled Cellule audit failures are retry POSTs after a matching GET, indicating +an availability gap for existing authenticated outcomes under pressure. The +summaries do not classify every failure by stage. Both Cellule owners exit zero +and log 1,000 Cell drains during cleanup; neither logs a fatal producer error. +That does not replace the missing controlled drain and cold audit. Celld owner +is OOM-killed, exit 137. Missing required after/cold provider observations also +fail qualification. None of these missing gates by itself proves mutation loss. + +The shared ARM64 Docker VM has 8 CPUs and 8 GiB total RAM. Node containers have +8-CPU/16-GiB ceilings and 4-GiB tmpfs; client/provider ceilings also exceed VM +resources. The runner supplies the 64-MiB retained-work and 1-GiB managed-disk +limits only to Cellule. Recording their values in celld case metadata does not +configure equivalent internal limits. This is workload-matched evidence with +asymmetric resource policies, not dedicated standard-node or device durability +qualification. Earlier descriptions of resource parity were too strong. + +## What still costs time + +Window-only canonical provider observations count all endpoint storage API +attempts, with successful in-window writes as the denominator. SDK-internal +retries and trailing publication are excluded; background work may cover other +command cohorts. + +| Observation | Baseline | Candidate | +| --- | ---: | ---: | +| GET/range attempts per completed write | 13.82 | 14.32 | +| PUT attempts per completed write | 0.7373 | 0.5212 | +| Materialized commands per selected root | 10.11 | 13.57 | +| Retained work at window end, MiB / 64 MiB | 58.39 | 27.62 | +| Oldest publication debt at window end, ms | 47,115 | 35,995 | +| Mean capture / worker / Fleet proof, ms | 0.241 / 2.268 / 7.912 | 0.262 / 2.595 / 7.678 | +| Mean Fleet response, ms | 464.37 | 367.36 | +| Mean background publication phase, ms | 59,825.96 | 63,587.41 | + +Different overlapping cohorts produce these timings; they cannot be added as +CPU cost. Background publication age includes delayed root scheduling and is +not command ACK latency. Bundle response counts are zero: Fleet wins the live +response race. The shared producer still supplies background coverage/cleanup. +The pressure fix is reproduced, but its fatal failure is absent from both fresh +application runs, so the completion difference cannot be isolated to that branch. + +Celld [bundles dirty Cell tails and pipelines ordered shipping](https://github.com/denoland/celld/blob/f2bf648663a610eefde71f3547ad61e9b896b1f0/crates/celld/ltx_repl.rs#L4411) +and [groups delivered follower appends before fsync](https://github.com/denoland/celld/blob/f2bf648663a610eefde71f3547ad61e9b896b1f0/crates/celld/node_log.rs#L160). +Cellule already uses managed SQLite WAL NORMAL, but still awaits one shipping +batch before the next and revalidates live historical/base dependencies during +selection. Bucket benchmark wiring bypasses the managed producer and publishes +per Cell. No new Bucket or read-only capacity measurement is made here. + +Next work must let existing read/retry outcomes remain available during pressure, +reserve materializer/cleanup progress before foreground admissions consume +credit, and reduce repeated dependency work while preserving exact authenticated +range, base and recovery contracts. Then measure ordered follower pipelining +and Bucket shared selection. Qualification still requires three matched +five-minute repetitions, zero errors/drops, all-ACK warm/cold recovery, bounded +debt, complete failed-owner suffix recovery, safe collection and read/mixed +profiles. The 2,000-Cell / 10K-write / 50K-read target remains unqualified. + +## Evidence + +Raw sources, binaries, failed attempts, journals, provider observations and +independent replay remain outside Git at +`/Volumes/Workspace/crabbuild-target/cellule-write-perf-8ad1` under +`receipt-admission-20261008-*`, plus the fresh baseline case in the retained +baseline build directory. Every offer/attempt/completion and complete ACK cohort +reconciles independently. See the external evidence index and verification +certificate for hashes and the complete retained inventory. The index contains +12,460 files / 1,843,123,095 bytes; every entry independently rehashes. Its SHA-256 +is `ede0f3083302324f1b4eee8a05bf501127eb156a7035d1ad981185f2e81c7974`. + +[Historical-read experiment](pr67-historical-read-measurement.md), +[bundle implementation](bundle-coverage-implementation.md), +[runtime design](../crates/cellule-runtime/docs/write-performance-design.md). From 607eca9ad844768fd7ee26cd66cbbec4cd8fad40 Mon Sep 17 00:00:00 2001 From: forhappy Date: Thu, 8 Oct 2026 21:54:31 -0700 Subject: [PATCH 056/102] Preserve durable command replays during publication pressure --- crates/cellule-runtime/docs/runtime.md | 8 +- .../docs/write-performance-design.md | 9 +- .../cellule-runtime/src/cell/actor/handle.rs | 19 +- .../src/cell/actor/lifecycle/scheduling.rs | 25 +- .../src/cell/actor/requests.rs | 23 ++ .../cellule-runtime/src/cell/actor/state.rs | 3 + .../src/node/bundle/tests/mod.rs | 1 + .../src/node/bundle/tests/readiness.rs | 10 +- .../src/node/bundle/tests/replay_pressure.rs | 293 ++++++++++++++++++ 9 files changed, 377 insertions(+), 14 deletions(-) create mode 100644 crates/cellule-runtime/src/node/bundle/tests/replay_pressure.rs diff --git a/crates/cellule-runtime/docs/runtime.md b/crates/cellule-runtime/docs/runtime.md index 9a5c2970..7e54e3a6 100644 --- a/crates/cellule-runtime/docs/runtime.md +++ b/crates/cellule-runtime/docs/runtime.md @@ -75,7 +75,13 @@ Caller cancellation does not remove accepted members or their drain obligations. While follower-proven work awaits object publication, new mutations are refused before SQL when retained RAM or local disk reaches three quarters of its node budget. The remaining headroom belongs to accepted work and publication. -Queries remain eligible, and a refused mutation has no new ledger outcome. +Queries remain eligible. A command refused by node publication pressure or a +full Cell publication queue can still replay its original durable request +outcome through the same bounded FIFO and read-only resolution path. An absent +or unproven outcome returns the original refusal without invoking its handler; +digest conflicts, result limits and identity expiry remain enforced. These +lookups retain mailbox, worker and node byte admission. A refused mutation has +no new ledger outcome and cannot join a mutating native group. Local LTX bodies remain charged to the disk budget; publication's RAM reservation covers shared encoder indexes, descriptor/path copies, and retained outcomes. The physical pending-byte high water and node-log coverage counters still count diff --git a/crates/cellule-runtime/docs/write-performance-design.md b/crates/cellule-runtime/docs/write-performance-design.md index 4418e4d7..76589f93 100644 --- a/crates/cellule-runtime/docs/write-performance-design.md +++ b/crates/cellule-runtime/docs/write-performance-design.md @@ -1,10 +1,13 @@ # Node write and read performance design -The latest [historical-read experiment](../../../docs/pr67-historical-read-measurement.md) +The [historical-read experiment](../../../docs/pr67-historical-read-measurement.md) is withdrawn: grouping historical requests passes its isolated regression but regresses Fleet completion from 201.82 to 18.18 writes/s and fails ACK availability. -Production code returns to `11843f6`. Reuse bounded working credit and reproduce -resource-ledger pressure before accepting another publication optimization. +The subsequent [receipt admission verification](../../../docs/pr67-receipt-admission-measurement.md) +fixes a reproduced producer pressure failure, but completes only 154.72 Fleet +writes/s and fails ACK availability. Resource policies in the celld comparison +are asymmetric. Reuse bounded working credit and qualify the original +application workload before accepting another publication optimization. Status: implementation in progress. The application path is not qualified at the targets below. Component I/O reductions are not application TPS. diff --git a/crates/cellule-runtime/src/cell/actor/handle.rs b/crates/cellule-runtime/src/cell/actor/handle.rs index 5a6a3ac6..778fe78c 100644 --- a/crates/cellule-runtime/src/cell/actor/handle.rs +++ b/crates/cellule-runtime/src/cell/actor/handle.rs @@ -214,12 +214,28 @@ impl CellHandle { } else { AdmissionKind::Command }; - let admission = self.reserve_work_kind(kind, operation_bytes, max_result_bytes)?; + let (admission, refused_mutation) = + match self.reserve_work_kind(kind, operation_bytes, max_result_bytes) { + Ok(admission) => (admission, None), + Err(error @ Error::Capacity("publication backlog")) => { + identity.validate(now_ms)?; + // Keep the original owner, FIFO, request and byte budgets. + // Resolution can replay an existing proof; it cannot admit SQL. + let admission = self.reserve_work_kind( + AdmissionKind::Resolve, + operation_bytes, + max_result_bytes, + )?; + (admission, Some(error)) + } + Err(error) => return Err(error), + }; let (reply, response) = oneshot::channel(); self.inner .sender .send(Message::Execute(Box::new(QueuedCommand { group: None, + refused_mutation, trace: tracing::debug_span!( target: "cellule_runtime::action", "cell_execution", @@ -275,6 +291,7 @@ impl CellHandle { .sender .send(Message::Execute(Box::new(QueuedCommand { group: None, + refused_mutation: None, trace: tracing::debug_span!( target: "cellule_runtime::action", "cell_effect_execution", diff --git a/crates/cellule-runtime/src/cell/actor/lifecycle/scheduling.rs b/crates/cellule-runtime/src/cell/actor/lifecycle/scheduling.rs index 4b8efc5f..ec1ffb7c 100644 --- a/crates/cellule-runtime/src/cell/actor/lifecycle/scheduling.rs +++ b/crates/cellule-runtime/src/cell/actor/lifecycle/scheduling.rs @@ -6,12 +6,24 @@ pub(in crate::cell::actor) fn schedule( active: &mut ActiveCell, lease_live: bool, ) -> CoordinationDecision { - let publication_blocked = active.queue.front().is_some_and(|work| { + let mut publication_blocked = active.queue.front().is_some_and(|work| { matches!(work, QueuedWork::Command(_)) && (super::super::materialization::blocks_commands(active) || active.coordination.publication_count() >= MAX_PENDING_PUBLICATIONS || active.publication_bytes >= PENDING_PUBLICATION_HIGH_WATER_BYTES) }); + if publication_blocked + && let Some(QueuedWork::Command(command)) = active.queue.front_mut() + && matches!(command.operation, QueuedOperation::Mutation { .. }) + { + // A full debt queue must not hide prior Fleet/bundle outcomes or hold + // every later query behind a mutation that cannot yet be admitted. + command + .refused_mutation + .get_or_insert(Error::PendingPublication); + command._work.kind = AdmissionKind::Resolve; + publication_blocked = false; + } active.coordination.step(CoordinationInput::Schedule { queue_empty: active.queue.is_empty(), publisher_ready: active.publisher.is_some(), @@ -38,7 +50,9 @@ pub(in crate::cell::actor) fn start_next( return; }; active.cancel_compaction_admission(); - if matches!(&work, QueuedWork::Command(_) | QueuedWork::Migration(_)) { + if matches!(&work, QueuedWork::Command(command) if command.refused_mutation.is_none()) + || matches!(&work, QueuedWork::Migration(_)) + { // Durable command outcomes, effects, Queue rows, and Workflow runs // remain release obligations until a fresh inventory proves otherwise. active.persisted_work = crate::primitives::maintenance::PersistedWorkInventory::unknown(); @@ -73,12 +87,15 @@ pub(in crate::cell::actor) fn start_next( let interrupt = active.interrupt.clone(); match work { QueuedWork::Command(mut command) => { - if matches!(command.operation, QueuedOperation::Mutation { .. }) { + if command.refused_mutation.is_none() + && matches!(command.operation, QueuedOperation::Mutation { .. }) + { let mut members = Vec::new(); while members.len() + 1 < MAX_NATIVE_GROUP && active.queue.front().is_some_and(|work| { matches!(work, QueuedWork::Command(next) - if matches!(next.operation, QueuedOperation::Mutation { .. })) + if next.refused_mutation.is_none() + && matches!(next.operation, QueuedOperation::Mutation { .. })) }) { let Some(QueuedWork::Command(next)) = active.queue.pop_front() else { diff --git a/crates/cellule-runtime/src/cell/actor/requests.rs b/crates/cellule-runtime/src/cell/actor/requests.rs index 455d3e66..cd5a410b 100644 --- a/crates/cellule-runtime/src/cell/actor/requests.rs +++ b/crates/cellule-runtime/src/cell/actor/requests.rs @@ -166,12 +166,35 @@ pub(super) async fn execute_command( let max_result_bytes = command.max_result_bytes; let worker_pool = pool.clone(); let worker_deadline = deadline.clone(); + let refused_mutation = command.refused_mutation.take(); let operation = async move { match queued_operation { QueuedOperation::Mutation { identity, operation_digest, } => { + if let Some(refusal) = refused_mutation { + identity.validate(now_ms)?; + drop(handler); + return match worker_pool + .resolve( + cell, + identity, + operation_digest, + now_ms, + max_result_bytes, + worker_deadline, + ) + .await? + { + Resolution::Committed(outcome) => { + Ok(WorkerExecution::Recorded(outcome)) + } + Resolution::Absent | Resolution::Unknown | Resolution::Expired => { + Err(refusal) + } + }; + } worker_pool .execute_until( cell, diff --git a/crates/cellule-runtime/src/cell/actor/state.rs b/crates/cellule-runtime/src/cell/actor/state.rs index c0d1e5ca..cf29bba4 100644 --- a/crates/cellule-runtime/src/cell/actor/state.rs +++ b/crates/cellule-runtime/src/cell/actor/state.rs @@ -162,6 +162,9 @@ pub(super) enum Message { pub(super) struct QueuedCommand { pub(super) group: Option, + // Pressure may admit only an original durable outcome lookup. An absent + // identity returns this refusal without invoking the mutation handler. + pub(super) refused_mutation: Option, pub(super) trace: tracing::Span, pub(super) telemetry: crate::fleet::telemetry::CellTelemetryHandle, pub(super) queued_at: std::time::Instant, diff --git a/crates/cellule-runtime/src/node/bundle/tests/mod.rs b/crates/cellule-runtime/src/node/bundle/tests/mod.rs index 24248889..ecd872ce 100644 --- a/crates/cellule-runtime/src/node/bundle/tests/mod.rs +++ b/crates/cellule-runtime/src/node/bundle/tests/mod.rs @@ -26,6 +26,7 @@ mod readiness; mod receipt_pressure; mod receipts; mod recovery; +mod replay_pressure; struct Fixture { count: Arc, layout: CellStorageLayout, diff --git a/crates/cellule-runtime/src/node/bundle/tests/readiness.rs b/crates/cellule-runtime/src/node/bundle/tests/readiness.rs index 3c9dd5b3..292f6d15 100644 --- a/crates/cellule-runtime/src/node/bundle/tests/readiness.rs +++ b/crates/cellule-runtime/src/node/bundle/tests/readiness.rs @@ -15,11 +15,11 @@ use futures_util::future::BoxFuture; use std::sync::atomic::{AtomicBool, AtomicUsize, Ordering}; use std::time::Duration; -struct DelayedSelection { - authority: Arc, - held: AtomicBool, - entered: tokio::sync::Notify, - changed: tokio::sync::Notify, +pub(super) struct DelayedSelection { + pub(super) authority: Arc, + pub(super) held: AtomicBool, + pub(super) entered: tokio::sync::Notify, + pub(super) changed: tokio::sync::Notify, } impl NodeBundleAuthority for DelayedSelection { diff --git a/crates/cellule-runtime/src/node/bundle/tests/replay_pressure.rs b/crates/cellule-runtime/src/node/bundle/tests/replay_pressure.rs new file mode 100644 index 00000000..22cd74e5 --- /dev/null +++ b/crates/cellule-runtime/src/node/bundle/tests/replay_pressure.rs @@ -0,0 +1,293 @@ +//! Original Fleet outcomes remain replayable when mutation debt is full. +use super::actor::{Authority, transport}; +use super::readiness::DelayedSelection; +use super::*; +use crate::cell::actor::CellRuntime; +use crate::cell::catalog::{CatalogEntry, CatalogRole, CellCatalog}; +use crate::cell::executor::{HandlerOutcome, MAX_PENDING_PUBLICATIONS, MutationIdentity}; +use crate::cell::worker::SqlWorkerPool; +use crate::identity::{CellTarget, NamespaceId, RequestId, TenantId}; +use crate::node::durability::NodeDurability; +use crate::node::log_shipper::NodeLogShipper; +use std::sync::atomic::{AtomicBool, Ordering}; +use std::time::Duration; + +#[tokio::test] +async fn replay_admission_retains_fleet_outcomes_while_new_writes_are_refused() { + pressure_case(true).await; +} + +#[tokio::test] +async fn replay_admission_retains_fleet_outcomes_at_the_cell_debt_limit() { + pressure_case(false).await; +} + +async fn pressure_case(node_pressure: bool) { + let mut f = Fixture::new().await; + super::coverage::enroll(&mut f).await; + let authority = Arc::new(Authority { + directory: f.directory.clone(), + observed: tokio::sync::Mutex::new(f.node.clone()), + }); + let delayed = Arc::new(DelayedSelection { + authority: authority.clone(), + held: AtomicBool::new(false), + entered: tokio::sync::Notify::new(), + changed: tokio::sync::Notify::new(), + }); + let peers = transport(&f, true); + let shipper = NodeLogShipper::new(f.gate.clone(), peers.clone(), Limits::default()).unwrap(); + let durability = Arc::new(NodeDurability::new( + f.gate.clone(), + shipper, + authority.clone(), + peers, + f.lease.clone(), + )); + let pool = SqlWorkerPool::new(2, 4).unwrap(); + let runtime = CellRuntime::new_with_replica_host_requiring_node_lease( + pool.clone(), + 64 << 20, + SessionId::from_bytes([1; 16]), + cellule_ltx::Host::default(), + ) + .unwrap(); + runtime.install_node_lease(f.lease.clone()).unwrap(); + runtime + .install_node_durability(ApplicationId::from_bytes([9; 16]), durability.clone()) + .unwrap(); + durability + .start_bundle_publication(delayed.clone()) + .unwrap(); + let target = CellTarget::new( + TenantId::from_bytes([1; 16]), + ApplicationId::from_bytes([9; 16]), + NamespaceId::from_bytes([13; 16]), + b"replay-pressure", + ) + .unwrap(); + let catalog = CellCatalog::new(f.layout.clone(), target.tenant()); + let proof = catalog + .provision( + CatalogEntry::new( + &target, + CatalogRole::Application, + Digest::from_bytes([12; 32]), + 1, + ) + .unwrap(), + ) + .await + .unwrap(); + let cell_authority = CellAuthority::new(f.layout.clone()); + let control = cell_authority + .create_initial( + &proof, + IncarnationId::from_bytes([4; 16]), + Owner { + session: SessionId::from_bytes([1; 16]), + endpoint: "https://bundle.internal:8081".into(), + }, + ) + .await + .unwrap(); + let replica = CellReplica::new( + f.layout.clone(), + *target.cell_id().as_bytes(), + [4; 16], + Limits::default(), + ) + .unwrap(); + let handle = runtime + .bootstrap( + proof, + replica.clone(), + cell_authority.clone(), + control, + f.scratch.path().join("pressure.sqlite"), + |tx| { + tx.execute_batch( + "CREATE TABLE counter(value INTEGER); INSERT INTO counter VALUES(0)", + )?; + Ok(()) + }, + ) + .await + .unwrap(); + delayed.held.store(true, Ordering::Release); + let identity = |byte| MutationIdentity { + request_id: RequestId::from_bytes([byte; 16]), + issued_at_ms: 10, + expires_at_ms: 10_000, + }; + let digest = Digest::from_bytes([9; 32]); + let first = handle + .execute(identity(1), digest, 20, 1024, 1024, |tx| { + tx.execute("UPDATE counter SET value=value+1", [])?; + Ok(HandlerOutcome::Success(b"ready".to_vec())) + }) + .await + .unwrap(); + tokio::time::timeout(Duration::from_secs(3), delayed.entered.notified()) + .await + .unwrap(); + // A real follower proof released this response; selection is still held. + assert_eq!( + f.gate.progress().unwrap().follower_proven_through, + f.gate.progress().unwrap().issued_through + ); + let resources = pool.resource_ledger(); + if node_pressure { + let snapshot = resources.snapshot().unwrap(); + let pressure = runtime + .try_reserve_node_metadata_bytes( + snapshot.limit.retained_bytes() * 3 / 4 - snapshot.used.retained_bytes(), + ) + .unwrap(); + let replay = tokio::time::timeout( + Duration::from_secs(3), + handle.execute(identity(1), digest, 21, 1024, 1024, |_| { + panic!("retry executed under node pressure") + }), + ) + .await + .unwrap() + .unwrap(); + assert_eq!(replay, first); + assert!(matches!( + handle + .execute(identity(200), digest, 21, 1024, 1024, |_| { + panic!("new mutation executed under node pressure") + }) + .await, + Err(Error::Capacity("publication backlog")) + )); + assert!(matches!( + handle + .execute( + identity(1), + Digest::from_bytes([8; 32]), + 21, + 1024, + 1024, + |_| { panic!("conflicting retry executed") } + ) + .await, + Err(Error::RequestConflict) + )); + assert!(matches!( + handle + .execute(identity(1), digest, 21, 1024, 1, |_| { + panic!("oversized retry executed") + }) + .await, + Err(Error::Command("stored result exceeds command limit")) + )); + assert!(matches!( + handle + .execute(identity(1), digest, 10_000, 1024, 1024, |_| { + panic!("expired retry executed") + }) + .await, + Err(Error::Command("invalid mutation identity lifetime")) + )); + drop(pressure); + } + // Independently fill the unchanged per-Cell physical debt limit. A FIFO + // retry at its head must not wait for root publication or block a query. + for byte in 2..=MAX_PENDING_PUBLICATIONS as u8 { + handle + .execute(identity(byte), digest, 21, 1024, 1024, |tx| { + tx.execute("UPDATE counter SET value=value+1", [])?; + Ok(HandlerOutcome::Success(b"ready".to_vec())) + }) + .await + .unwrap(); + } + assert_eq!( + runtime + .publication_progress() + .await + .unwrap() + .pending_publications, + MAX_PENDING_PUBLICATIONS + ); + assert_eq!( + tokio::time::timeout( + Duration::from_secs(3), + handle.execute(identity(1), digest, 22, 1024, 1024, |_| panic!( + "retry executed under Cell pressure" + ),) + ) + .await + .unwrap() + .unwrap(), + first + ); + assert!(matches!( + tokio::time::timeout( + Duration::from_secs(3), + handle.execute(identity(201), digest, 22, 1024, 1024, |_| panic!( + "new mutation executed under Cell pressure" + ),) + ) + .await + .unwrap(), + Err(Error::PendingPublication) + )); + assert!(matches!( + handle + .execute(identity(1), digest, 10_000, 1024, 1024, |_| { + panic!("expired retry executed at the Cell debt limit") + }) + .await, + Err(Error::Command("invalid mutation identity lifetime")) + )); + assert_eq!( + handle + .query(1024, 1024, |db| Ok(db + .query_row("SELECT value FROM counter", [], |row| row.get::<_, i64>(0),)? + .to_le_bytes() + .to_vec())) + .await + .unwrap(), + (MAX_PENDING_PUBLICATIONS as i64).to_le_bytes() + ); + delayed.held.store(false, Ordering::Release); + delayed.changed.notify_waiters(); + super::managed::renew_actor_lease(&authority, &f.lease).await; + tokio::time::timeout(Duration::from_secs(15), runtime.shutdown()) + .await + .unwrap() + .unwrap(); + let control = cell_authority + .load(target.cell_id()) + .await + .unwrap() + .unwrap(); + assert_eq!(control.value().state, ControlState::Idle); + assert!(control.value().bundle_binding.is_none()); + let root = control.value().ltx_root().unwrap(); + assert_eq!(root.commit_sequence, MAX_PENDING_PUBLICATIONS as u64); + let path = f.scratch.path().join("cold-pressure.sqlite"); + replica + .open_root(&root) + .await + .unwrap() + .restore(&path) + .await + .unwrap(); + let cold = cellule_ltx::rusqlite::Connection::open(path).unwrap(); + assert_eq!( + cold.query_row("SELECT value FROM counter", [], |row| row.get::<_, i64>(0)) + .unwrap(), + MAX_PENDING_PUBLICATIONS as i64 + ); + assert_eq!( + cold.query_row("SELECT COUNT(*) FROM sys_requests", [], |row| row + .get::<_, i64>(0)) + .unwrap(), + MAX_PENDING_PUBLICATIONS as i64 + ); + assert_eq!(resources.snapshot().unwrap().used.retained_bytes(), 0); +} From c45693aa6c79cb8c83a0dedbd2db0ad9977934ac Mon Sep 17 00:00:00 2001 From: forhappy Date: Thu, 8 Oct 2026 22:11:04 -0700 Subject: [PATCH 057/102] Retain admitted mutations after a pressure outcome probe --- crates/cellule-runtime/docs/runtime.md | 10 +- .../src/cell/actor/admission.rs | 1 + .../cellule-runtime/src/cell/actor/handle.rs | 4 + .../src/cell/actor/lifecycle/scheduling.rs | 23 ++-- crates/cellule-runtime/src/cell/actor/mod.rs | 1 + .../cellule-runtime/src/cell/actor/replay.rs | 104 ++++++++++++++++++ .../src/cell/actor/requests.rs | 26 +---- .../cellule-runtime/src/cell/actor/state.rs | 3 + .../src/cell/actor/tasks/work.rs | 6 + .../src/node/bundle/tests/replay_pressure.rs | 40 ++++--- 10 files changed, 169 insertions(+), 49 deletions(-) create mode 100644 crates/cellule-runtime/src/cell/actor/replay.rs diff --git a/crates/cellule-runtime/docs/runtime.md b/crates/cellule-runtime/docs/runtime.md index 7e54e3a6..40196c53 100644 --- a/crates/cellule-runtime/docs/runtime.md +++ b/crates/cellule-runtime/docs/runtime.md @@ -77,11 +77,13 @@ before SQL when retained RAM or local disk reaches three quarters of its node budget. The remaining headroom belongs to accepted work and publication. Queries remain eligible. A command refused by node publication pressure or a full Cell publication queue can still replay its original durable request -outcome through the same bounded FIFO and read-only resolution path. An absent -or unproven outcome returns the original refusal without invoking its handler; -digest conflicts, result limits and identity expiry remain enforced. These +outcome through the same bounded FIFO and read-only resolution path. Node +pressure keeps an absent or unproven outcome refused without invoking its handler. +At a full Cell queue, an absent request keeps its already accepted FIFO position +and handler until publication frees capacity; it is probed only once while blocked. +Digest conflicts, result limits and identity expiry remain enforced. These lookups retain mailbox, worker and node byte admission. A refused mutation has -no new ledger outcome and cannot join a mutating native group. +no new ledger outcome. Read-only outcome probes cannot join a mutating native group. Local LTX bodies remain charged to the disk budget; publication's RAM reservation covers shared encoder indexes, descriptor/path copies, and retained outcomes. The physical pending-byte high water and node-log coverage counters still count diff --git a/crates/cellule-runtime/src/cell/actor/admission.rs b/crates/cellule-runtime/src/cell/actor/admission.rs index a001b906..30e3d839 100644 --- a/crates/cellule-runtime/src/cell/actor/admission.rs +++ b/crates/cellule-runtime/src/cell/actor/admission.rs @@ -181,6 +181,7 @@ pub(super) fn send_command_task_reply( return; } Ok(CommandTaskResult::Recorded(outcome)) => Ok(outcome), + Ok(CommandTaskResult::AwaitPublication) => Err(Error::Fenced), Ok(CommandTaskResult::Pending { .. }) => Err(command.operation.unknown(Error::Fenced)), Err(error) => Err(error), }; diff --git a/crates/cellule-runtime/src/cell/actor/handle.rs b/crates/cellule-runtime/src/cell/actor/handle.rs index 778fe78c..128d357e 100644 --- a/crates/cellule-runtime/src/cell/actor/handle.rs +++ b/crates/cellule-runtime/src/cell/actor/handle.rs @@ -236,6 +236,8 @@ impl CellHandle { .send(Message::Execute(Box::new(QueuedCommand { group: None, refused_mutation, + publication_probe: false, + publication_probed: false, trace: tracing::debug_span!( target: "cellule_runtime::action", "cell_execution", @@ -292,6 +294,8 @@ impl CellHandle { .send(Message::Execute(Box::new(QueuedCommand { group: None, refused_mutation: None, + publication_probe: false, + publication_probed: false, trace: tracing::debug_span!( target: "cellule_runtime::action", "cell_effect_execution", diff --git a/crates/cellule-runtime/src/cell/actor/lifecycle/scheduling.rs b/crates/cellule-runtime/src/cell/actor/lifecycle/scheduling.rs index ec1ffb7c..d3e8923f 100644 --- a/crates/cellule-runtime/src/cell/actor/lifecycle/scheduling.rs +++ b/crates/cellule-runtime/src/cell/actor/lifecycle/scheduling.rs @@ -16,13 +16,17 @@ pub(in crate::cell::actor) fn schedule( && let Some(QueuedWork::Command(command)) = active.queue.front_mut() && matches!(command.operation, QueuedOperation::Mutation { .. }) { - // A full debt queue must not hide prior Fleet/bundle outcomes or hold - // every later query behind a mutation that cannot yet be admitted. - command - .refused_mutation - .get_or_insert(Error::PendingPublication); - command._work.kind = AdmissionKind::Resolve; - publication_blocked = false; + // Probe once for an original durable result. A missing identity retains + // its accepted FIFO position and handler until debt capacity returns. + if command.refused_mutation.is_some() { + publication_blocked = false; + } else if !command.publication_probed { + command.publication_probe = true; + publication_blocked = false; + } + } else if let Some(QueuedWork::Command(command)) = active.queue.front_mut() { + command.publication_probe = false; + command.publication_probed = false; } active.coordination.step(CoordinationInput::Schedule { queue_empty: active.queue.is_empty(), @@ -50,7 +54,8 @@ pub(in crate::cell::actor) fn start_next( return; }; active.cancel_compaction_admission(); - if matches!(&work, QueuedWork::Command(command) if command.refused_mutation.is_none()) + if matches!(&work, QueuedWork::Command(command) + if command.refused_mutation.is_none() && !command.publication_probe) || matches!(&work, QueuedWork::Migration(_)) { // Durable command outcomes, effects, Queue rows, and Workflow runs @@ -88,6 +93,7 @@ pub(in crate::cell::actor) fn start_next( match work { QueuedWork::Command(mut command) => { if command.refused_mutation.is_none() + && !command.publication_probe && matches!(command.operation, QueuedOperation::Mutation { .. }) { let mut members = Vec::new(); @@ -95,6 +101,7 @@ pub(in crate::cell::actor) fn start_next( && active.queue.front().is_some_and(|work| { matches!(work, QueuedWork::Command(next) if next.refused_mutation.is_none() + && !next.publication_probe && matches!(next.operation, QueuedOperation::Mutation { .. })) }) { diff --git a/crates/cellule-runtime/src/cell/actor/mod.rs b/crates/cellule-runtime/src/cell/actor/mod.rs index 70325f9d..1a9ff280 100644 --- a/crates/cellule-runtime/src/cell/actor/mod.rs +++ b/crates/cellule-runtime/src/cell/actor/mod.rs @@ -35,6 +35,7 @@ mod maintenance; mod materialization; pub use maintenance::MaintenanceCellRelease; mod receiver; +mod replay; mod requests; pub(crate) mod routes; mod runtime; diff --git a/crates/cellule-runtime/src/cell/actor/replay.rs b/crates/cellule-runtime/src/cell/actor/replay.rs new file mode 100644 index 00000000..b282a675 --- /dev/null +++ b/crates/cellule-runtime/src/cell/actor/replay.rs @@ -0,0 +1,104 @@ +//! Read-only command lookup under publication pressure, using the original gate. +use super::admission::{fence_admission, send_command_reply}; +use super::*; +use tracing::Instrument as _; + +pub(super) async fn execute( + pool: SqlWorkerPool, + mut command: Box, + interrupt: Arc, + generation: u64, + effect_id: u64, +) -> TaskResult { + let started = std::time::Instant::now(); + let queue_wait = command.queued_at.elapsed(); + let deadline = SqlDeadline::new(started + SQL_WALL_DEADLINE); + let operation = command.operation; + let cell = command.cell; + let now_ms = command.now_ms; + let max_result_bytes = command.max_result_bytes; + let worker_pool = pool.clone(); + let worker_deadline = deadline.clone(); + let lookup = async move { + let QueuedOperation::Mutation { + identity, + operation_digest, + } = operation + else { + return Err(Error::Control("publication lookup contains an effect")); + }; + identity.validate(now_ms)?; + worker_pool + .resolve( + cell, + identity, + operation_digest, + now_ms, + max_result_bytes, + worker_deadline, + ) + .await + }; + let lookup = lookup.instrument(command.trace.clone()); + tokio::pin!(lookup); + let resolution = match tokio::time::timeout_at(deadline.at().into(), &mut lookup).await { + Ok(result) => result, + Err(_) => { + let fenced = !deadline.cancel_queued(); + if fenced { + interrupt.interrupt(); + fence_admission(&command.admission); + } + let error = if fenced { + command.operation.unknown(Error::Deadline) + } else { + Error::Deadline + }; + send_command_reply(&mut command, Err(error)); + // The original command admission survives cancellation until its + // dispatched read exits, just as it does for normal command SQL. + let _ = lookup.as_mut().await; + if fenced { + let _ = pool.fence(command.cell).await; + } + Err(Error::Deadline) + } + }; + let result = match resolution { + Ok(Resolution::Committed(outcome)) => Ok(CommandTaskResult::Recorded(outcome)), + Ok(Resolution::Absent) if command.refused_mutation.is_none() => { + Ok(CommandTaskResult::AwaitPublication) + } + Ok(Resolution::Absent | Resolution::Unknown) => Err(command + .refused_mutation + .take() + .unwrap_or(Error::PendingPublication)), + Ok(Resolution::Expired) => Err(Error::Command("invalid mutation identity lifetime")), + Err(error) => Err(error), + }; + command + .telemetry + .command_execution(queue_wait, started.elapsed(), result.is_ok()); + let fenced = result.is_err() + && !deadline.cancelled() + && !pool + .state(command.cell) + .await + .is_ok_and(WorkerState::is_reusable); + if fenced { + let _ = pool.fence(command.cell).await; + } + let result = if fenced { + result.map_err(|error| command.operation.unknown(error)) + } else { + result + }; + TaskResult::Executed { + cell: command.cell, + generation, + effect_id, + command, + result, + fenced, + } +} diff --git a/crates/cellule-runtime/src/cell/actor/requests.rs b/crates/cellule-runtime/src/cell/actor/requests.rs index cd5a410b..4b20e17a 100644 --- a/crates/cellule-runtime/src/cell/actor/requests.rs +++ b/crates/cellule-runtime/src/cell/actor/requests.rs @@ -146,6 +146,9 @@ pub(super) async fn execute_command( effect_id: u64, ) -> TaskResult { let execution_started = std::time::Instant::now(); + if command.refused_mutation.is_some() || command.publication_probe { + return super::replay::execute(pool, command, interrupt, generation, effect_id).await; + } if command.group.is_some() { return super::group::execute(pool, durability, command, interrupt, generation, effect_id) .await; @@ -166,35 +169,12 @@ pub(super) async fn execute_command( let max_result_bytes = command.max_result_bytes; let worker_pool = pool.clone(); let worker_deadline = deadline.clone(); - let refused_mutation = command.refused_mutation.take(); let operation = async move { match queued_operation { QueuedOperation::Mutation { identity, operation_digest, } => { - if let Some(refusal) = refused_mutation { - identity.validate(now_ms)?; - drop(handler); - return match worker_pool - .resolve( - cell, - identity, - operation_digest, - now_ms, - max_result_bytes, - worker_deadline, - ) - .await? - { - Resolution::Committed(outcome) => { - Ok(WorkerExecution::Recorded(outcome)) - } - Resolution::Absent | Resolution::Unknown | Resolution::Expired => { - Err(refusal) - } - }; - } worker_pool .execute_until( cell, diff --git a/crates/cellule-runtime/src/cell/actor/state.rs b/crates/cellule-runtime/src/cell/actor/state.rs index cf29bba4..68986d43 100644 --- a/crates/cellule-runtime/src/cell/actor/state.rs +++ b/crates/cellule-runtime/src/cell/actor/state.rs @@ -165,6 +165,8 @@ pub(super) struct QueuedCommand { // Pressure may admit only an original durable outcome lookup. An absent // identity returns this refusal without invoking the mutation handler. pub(super) refused_mutation: Option, + pub(super) publication_probe: bool, + pub(super) publication_probed: bool, pub(super) trace: tracing::Span, pub(super) telemetry: crate::fleet::telemetry::CellTelemetryHandle, pub(super) queued_at: std::time::Instant, @@ -638,6 +640,7 @@ pub(super) enum TaskResult { pub(super) enum CommandTaskResult { Recorded(StoredOutcome), + AwaitPublication, GroupRecorded, Pending { pending: Box, diff --git a/crates/cellule-runtime/src/cell/actor/tasks/work.rs b/crates/cellule-runtime/src/cell/actor/tasks/work.rs index 05e0a7f7..8804a4e0 100644 --- a/crates/cellule-runtime/src/cell/actor/tasks/work.rs +++ b/crates/cellule-runtime/src/cell/actor/tasks/work.rs @@ -47,6 +47,12 @@ pub(super) fn handle_executed( return; } match result { + Ok(CommandTaskResult::AwaitPublication) => { + finish_work(active, false); + command.publication_probe = false; + command.publication_probed = true; + active.queue.push_front(QueuedWork::Command(command)); + } Ok(CommandTaskResult::GroupRecorded) => { finish_work(active, false); release_command_request_slots(&mut command); diff --git a/crates/cellule-runtime/src/node/bundle/tests/replay_pressure.rs b/crates/cellule-runtime/src/node/bundle/tests/replay_pressure.rs index 22cd74e5..f49bd958 100644 --- a/crates/cellule-runtime/src/node/bundle/tests/replay_pressure.rs +++ b/crates/cellule-runtime/src/node/bundle/tests/replay_pressure.rs @@ -224,17 +224,6 @@ async fn pressure_case(node_pressure: bool) { .unwrap(), first ); - assert!(matches!( - tokio::time::timeout( - Duration::from_secs(3), - handle.execute(identity(201), digest, 22, 1024, 1024, |_| panic!( - "new mutation executed under Cell pressure" - ),) - ) - .await - .unwrap(), - Err(Error::PendingPublication) - )); assert!(matches!( handle .execute(identity(1), digest, 10_000, 1024, 1024, |_| { @@ -253,8 +242,31 @@ async fn pressure_case(node_pressure: bool) { .unwrap(), (MAX_PENDING_PUBLICATIONS as i64).to_le_bytes() ); + let ran = Arc::new(AtomicBool::new(false)); + let executed = ran.clone(); + let fresh = handle.execute(identity(201), digest, 22, 1024, 1024, move |tx| { + assert!(!executed.swap(true, Ordering::SeqCst)); + tx.execute("UPDATE counter SET value=value+1", [])?; + Ok(HandlerOutcome::Success(b"accepted".to_vec())) + }); + tokio::pin!(fresh); + assert!( + tokio::time::timeout(Duration::from_millis(100), fresh.as_mut()) + .await + .is_err() + ); + assert!(!ran.load(Ordering::SeqCst)); delayed.held.store(false, Ordering::Release); delayed.changed.notify_waiters(); + let accepted = tokio::time::timeout(Duration::from_secs(5), fresh.as_mut()) + .await + .unwrap() + .unwrap(); + assert_eq!( + accepted.commit_sequence(), + MAX_PENDING_PUBLICATIONS as u64 + 1 + ); + assert!(ran.load(Ordering::SeqCst)); super::managed::renew_actor_lease(&authority, &f.lease).await; tokio::time::timeout(Duration::from_secs(15), runtime.shutdown()) .await @@ -268,7 +280,7 @@ async fn pressure_case(node_pressure: bool) { assert_eq!(control.value().state, ControlState::Idle); assert!(control.value().bundle_binding.is_none()); let root = control.value().ltx_root().unwrap(); - assert_eq!(root.commit_sequence, MAX_PENDING_PUBLICATIONS as u64); + assert_eq!(root.commit_sequence, MAX_PENDING_PUBLICATIONS as u64 + 1); let path = f.scratch.path().join("cold-pressure.sqlite"); replica .open_root(&root) @@ -281,13 +293,13 @@ async fn pressure_case(node_pressure: bool) { assert_eq!( cold.query_row("SELECT value FROM counter", [], |row| row.get::<_, i64>(0)) .unwrap(), - MAX_PENDING_PUBLICATIONS as i64 + MAX_PENDING_PUBLICATIONS as i64 + 1 ); assert_eq!( cold.query_row("SELECT COUNT(*) FROM sys_requests", [], |row| row .get::<_, i64>(0)) .unwrap(), - MAX_PENDING_PUBLICATIONS as i64 + MAX_PENDING_PUBLICATIONS as i64 + 1 ); assert_eq!(resources.snapshot().unwrap().used.retained_bytes(), 0); } From 9f2a39d6726470a48a14b92b109f896cc0977a46 Mon Sep 17 00:00:00 2001 From: forhappy Date: Thu, 8 Oct 2026 22:43:30 -0700 Subject: [PATCH 058/102] Record durable-replay verification and publication admission gap --- crates/cellule-runtime/docs/runtime.md | 4 +- .../docs/write-performance-design.md | 19 +- docs/bundle-coverage-implementation.md | 16 +- docs/pr67-replay-admission-measurement.md | 165 ++++++++++++++++++ 4 files changed, 187 insertions(+), 17 deletions(-) create mode 100644 docs/pr67-replay-admission-measurement.md diff --git a/crates/cellule-runtime/docs/runtime.md b/crates/cellule-runtime/docs/runtime.md index 40196c53..5ed43424 100644 --- a/crates/cellule-runtime/docs/runtime.md +++ b/crates/cellule-runtime/docs/runtime.md @@ -75,8 +75,8 @@ Caller cancellation does not remove accepted members or their drain obligations. While follower-proven work awaits object publication, new mutations are refused before SQL when retained RAM or local disk reaches three quarters of its node budget. The remaining headroom belongs to accepted work and publication. -Queries remain eligible. A command refused by node publication pressure or a -full Cell publication queue can still replay its original durable request +Queries remain eligible. Under node publication pressure or a full Cell +publication queue, a command can still replay its original durable request outcome through the same bounded FIFO and read-only resolution path. Node pressure keeps an absent or unproven outcome refused without invoking its handler. At a full Cell queue, an absent request keeps its already accepted FIFO position diff --git a/crates/cellule-runtime/docs/write-performance-design.md b/crates/cellule-runtime/docs/write-performance-design.md index 76589f93..de37ae5f 100644 --- a/crates/cellule-runtime/docs/write-performance-design.md +++ b/crates/cellule-runtime/docs/write-performance-design.md @@ -1,13 +1,16 @@ # Node write and read performance design -The [historical-read experiment](../../../docs/pr67-historical-read-measurement.md) -is withdrawn: grouping historical requests passes its isolated regression but -regresses Fleet completion from 201.82 to 18.18 writes/s and fails ACK availability. -The subsequent [receipt admission verification](../../../docs/pr67-receipt-admission-measurement.md) -fixes a reproduced producer pressure failure, but completes only 154.72 Fleet -writes/s and fails ACK availability. Resource policies in the celld comparison -are asymmetric. Reuse bounded working credit and qualify the original -application workload before accepting another publication optimization. +The [latest replay-pressure verification](../../../docs/pr67-replay-admission-measurement.md) +preserves original durable outcomes under publication pressure. One fresh Fleet +pair completes 158.15 writes/s versus 152.92 before, with all 19,542 candidate ACKs +passing warm/cold audit and joined drain. A separate 1-GiB budget control reaches +only 191.08 writes/s; its latency worsens versus the matching baseline. Neither +establishes repeatable performance improvement or parity. Native ticket issuance +still waits for the bounded publication lane before follower proof can begin, +and historical/base verification remains expensive. Resource policies in the +celld comparison are asymmetric. The +[historical-read experiment](../../../docs/pr67-historical-read-measurement.md) +remains withdrawn after its severe application regression. Status: implementation in progress. The application path is not qualified at the targets below. Component I/O reductions are not application TPS. diff --git a/docs/bundle-coverage-implementation.md b/docs/bundle-coverage-implementation.md index 4bece97b..30dbab0f 100644 --- a/docs/bundle-coverage-implementation.md +++ b/docs/bundle-coverage-implementation.md @@ -1,17 +1,19 @@ # Bundle coverage implementation -The [latest receipt-pressure diagnostic](pr67-receipt-admission-measurement.md) -reproduces a fatal post-selection admission failure and verifies waiting for -credit while joining older checkpoints. One fresh Fleet pair completes 154.72 -writes/s versus 122.97 before and 4,001.65 for celld. All three fail warm ACK -availability and qualification; resource policies are asymmetric. No repeatable +The [latest replay-pressure diagnostic](pr67-replay-admission-measurement.md) +preserves original durable command outcomes under node and Cell publication +pressure. One fresh Fleet pair completes 158.15 writes/s versus 152.92 before +and 4,473.32 for celld. Candidate warm/cold audits pass for all 19,542 ACKs; +all three fail performance qualification. A separate 1-GiB retained-credit control +reaches only 191.08 writes/s. Resource policies are asymmetric, and no repeatable throughput improvement or parity is established. PR #67 remains a draft. The installed original producer provides admitted shared receipts, independent root materialization and complete live-writer closure. The Fleet SQL example installs it; Bucket benchmark wiring bypasses it. Remaining gaps are publication -and verification cost, pressure-safe read/retry and materializer progress, -ordered shipping, safe collection and full lifecycle/performance qualification. +and verification cost, publication-coupled native admission, materializer +progress and remaining overload availability, ordered shipping, safe collection +and full lifecycle/performance qualification. The [historical-read experiment](pr67-historical-read-measurement.md) was reverted after a severe Fleet regression. Earlier measured slices remain linked from the [delivery record](write-performance-delivery.md). diff --git a/docs/pr67-replay-admission-measurement.md b/docs/pr67-replay-admission-measurement.md new file mode 100644 index 00000000..36689162 --- /dev/null +++ b/docs/pr67-replay-admission-measurement.md @@ -0,0 +1,165 @@ +# PR 67: durable replays and publication admission + +**Performance parity remains unmet.** Candidate +`c45693aa6c79cb8c83a0dedbd2db0ad9977934ac` completes 158.15 Fleet writes/s +versus 152.92 before and 4,473.32 for celld in fresh overloaded diagnostics. +The candidate passes complete warm/cold ACK read and retry audits; the baseline +fails its warm audit. A separate 1-GiB budget control reaches only 191.08 +writes/s. Raising retained credit alone does not close the throughput gap. + +## Reproduced failure and delivered change + +Previously, node publication pressure refused a command before checking whether +its original outcome was already durable. A full Cell publication queue also +left the same retry waiting for object selection. Two real actor/Fleet tests +reproduce these failures in three baseline executions: capacity refusal at the +node limit and timeout at the Cell limit. + +The candidate uses the existing bounded mailbox, worker and node byte admission +for a read-only lookup of that original request. The normal reply gate still +checks the original owner and durability visibility. Digest conflicts, result +limits and identity expiry remain enforced. An absent node-pressure request +retains its original refusal and never invokes its handler. An already admitted +fresh request at a full Cell queue is probed once, then retains its handler and +original FIFO position until capacity returns. Probes cannot join mutating native +groups or invalidate persisted-work inventory. + +The initial implementation rejected that already admitted fresh request. The +existing full-suite FIFO contract caught the regression. The final implementation +preserves the contract without changing its test or any qualification gate. +The two new regressions pass three times after correction, including exact +original outcomes, cold counter/ledger state, joined closure and zero retained +credit. Inbox-effect delivery is not changed by this command-replay fix. + +All contributor checks and Rust 1.99 Clippy pass in an immutable final snapshot: +1,965 workspace tests and 60 local LTX tests pass; 38 environment-dependent tests +remain ignored. All 1,300 Rust/Cargo files match the verified snapshot and pinned +production source used for the Linux build. Raw failed builds, the intermediate +FIFO failure and final verification are retained outside Git. + +## Fresh original-budget comparison + +Baseline: `addc1cecefb42a094bc6a1c42289d43334c0e645`. +Celld: `f2bf648663a610eefde71f3547ad61e9b896b1f0`. +The three sequential cases use identical client/auditor binaries, fixtures and +images: 1,000 uniform Cells, 96-byte SQL values, INSERT plus SELECT and a two-hour +outcome ledger, 128 clients/queue slots, 30-second warmup, 60-second window and +15K offered writes/s. No build or contributor suite overlaps a timed window. + +| System | Completed writes/s | Successful scheduled p99 ms | Successful request p99 ms | Errors | Queue drops | +| --- | ---: | ---: | ---: | ---: | ---: | +| Cellule baseline | 152.92 | 2,595.27 | 2,211.48 | 609,543 | 281,244 | +| Cellule candidate | 158.15 | 1,761.20 | 1,097.45 | 367,845 | 522,434 | +| celld | 4,473.32 | 95.22 | 51.03 | 5,447 | 626,152 | + +TPS counts successes inside the 60-second window. Successful p99 also includes +trailing measured successes and excludes errors/drops; scheduled latency begins +at offered arrival. Qualification retains all-attempt latency and zero-drop +requirements. Candidate has 9,489 successes inside and 232 after the window. +The 3.42% completion difference in one overloaded pair is not a repeatable gain. +Candidate also records 7,385 warmup request errors versus zero in the baseline; +warmup drops and attempt populations differ. + +| System | Complete ACK cohort | Warm errors / completed retries | Cold audit | Controlled drain | +| --- | ---: | ---: | --- | --- | +| Cellule baseline | 18,647 | 4,305 / 14,342 | not reached | unverified | +| Cellule candidate | 19,542 | 0 / 19,542 | 19,542 reads and retries; zero errors | 30.03 s | +| celld | 465,076 | 458,655 / 6,421 | not reached | unverified | + +Candidate provider health/lifecycle observations also pass. The baseline's +sampled warm failures are retry POSTs returning 503. Celld's sampled failures +are GETs returning 500, and its owner is OOM-killed, exit 137. Failed warm audits +prevent controlled drain/cold qualification and required after/cold observations; +missing gates do not themselves prove mutation loss. Every offer, attempt, +completion and complete ACK cohort reconciles independently. + +The shared ARM64 Docker VM has 8 CPUs and 8 GiB total RAM. Node container ceilings +are 8 CPUs/16 GiB with 4-GiB tmpfs; client/provider ceilings also exceed the VM. +Only Cellule receives the explicit 64-MiB retained-work and 1-GiB managed-disk +limits. Celld metadata records these values without applying equivalent limits. +These are workload-matched diagnostics with asymmetric resource policies, +not dedicated standard-node or physical-device qualification. All runs fail +performance qualification; the completed candidate lifecycle is not a passing +capacity profile. + +## Budget control + +Two additional sequential Cellule cases change only retained-work credit from +64 MiB to 1 GiB, using the same before/after binaries, workload, VM, client, +auditor and 1-GiB managed-disk limit. This is a separately labelled diagnostic, +not a replacement for the original profile or a matched celld qualification. + +| Code | Completed writes/s | Successful scheduled p99 ms | Successful request p99 ms | Errors | Queue drops | Warm/cold ACK cohort | Drain s | +| --- | ---: | ---: | ---: | ---: | ---: | ---: | ---: | +| Baseline | 170.42 | 2,475.05 | 1,472.67 | 0 | 889,519 | 20,855; zero errors | 26.39 | +| Candidate | 191.08 | 2,613.81 | 1,782.00 | 0 | 888,279 | 21,871; zero errors | 23.02 | + +The candidate's observed completion difference is 12.13%, but scheduled and +request p99 worsen. Neither row meets delivery or latency requirements. The +larger-budget baseline ends with only 27.34 MiB retained and 209.28 MiB of managed +disk reserved, while median request latency is 692.2 ms. Ample credit does not +remove the slow path. The three-arm comparator correctly rejects this two-arm +control; that rejected invocation is retained. Both rows independently reconcile +and their before/after workload and execution identities match. + +## Architectural gap and next measurable work + +Shared SQLite, LTX, node bundles and follower logs are components, not an equal +critical path. Managed Cellule already uses WAL NORMAL. In the larger-budget +baseline, mean native capture is 0.196 ms, worker execution 1.091 ms, actor queue +29.039 ms, Fleet proof 7.330 ms and Fleet response 732.140 ms. These are different +overlapping cohorts, not an additive per-request latency decomposition. + +One important wait is outside the worker/proof timers: +[`execute_command`](../crates/cellule-runtime/src/cell/actor/requests.rs) records +worker execution, then awaits durability submission. +[`submit_capture`](../crates/cellule-runtime/src/node/log_shipper/mod.rs) reserves +native bytes and a shipping slot, then holds the ordered issuance lock while +waiting for publication queue capacity **before** committing the ticket. +The [publication feed](../crates/cellule-runtime/src/node/log_shipper/publication/mod.rs) +has 512 submissions and retains shared outstanding-byte admission through joined +selection. Consequently, slow object selection can throttle Fleet admission +before the follower proof timer starts. A larger node RAM ledger does not change +this separate bounded lane. Per-stage wait measurements are still needed to +quantify its share of the total gap. + +Cellule [selection](../crates/cellule-runtime/src/node/bundle/selection.rs) +awaits each affected binding's verification in turn. The +[verifier](../crates/cellule-runtime/src/node/bundle/proof.rs) reopens the base and +historical ranges, even though the new cohort is read once. Native shipping also +awaits each complete append batch before beginning the next. Celld +[pipelines ordered shipping rounds and bundles dirty Cell tails](https://github.com/denoland/celld/blob/f2bf648663a610eefde71f3547ad61e9b896b1f0/crates/celld/ltx_repl.rs#L4530), +and [groups already delivered follower appends before fsync](https://github.com/denoland/celld/blob/f2bf648663a610eefde71f3547ad61e9b896b1f0/crates/celld/node_log.rs#L160). + +Original-budget candidate window observations are 13.64 GET/range attempts and +0.6106 successful PUTs per completed write, with 11.24 commands per materialized +root. This is still expensive background work. Endpoint API counters include +background cohorts; the denominator includes only successful in-window writes. +SDK retries and trailing publication are excluded. Fleet wins every observed +response race; zero Bundle responses do not mean the shared producer is disabled. +Candidate worker timings include 377,611 executions, mostly read-only pressure +probes, versus 9,617 captures. Their smaller average is not faster mutation SQL. + +| Next slice | Required evidence before accepting it | +| --- | --- | +| Instrument post-SQL submission | Same-request timestamps for native-byte wait, shipping slot, ordered lock, publication slot, load/encode, ticket issuance, proof and final reply; reconcile bounded occupancy | +| Reduce repeated selection work | Original-authority retention and exact range witnesses; verify new bytes once without skipping cold dependencies; lower total GET/PUT cost in paired application runs | +| Separate native progress from publication debt | Recoverable bounded backlog before issuance; all prior Fleet ACKs survive complete-range recovery and drain; queue enlargement alone is not sustainable throughput | +| Pipeline follower rounds | Ordered confirmation and durable grouped appends; no release past an unresolved earlier range; fault and complete-suffix recovery tests | +| Qualify the node | Three matched five-minute repetitions, zero errors/drops, bounded debt, all-ACK warm/cold recovery, safe collection and read/mixed guardrails at the unchanged targets | + +No new read-only or Bucket capacity result is claimed. The 2,000-Cell / +10K-write / 50K-read target and conditional 215-command checkpoint cost target +remain unqualified. PR #67 remains a draft. + +## Evidence + +Raw sources, binaries, fixtures, failed attempts, verification, journals, +provider observations and independent replay remain outside Git under +`/Volumes/Workspace/crabbuild-target/cellule-write-perf-8ad1/replay-admission-20261008-*`. +The external evidence index records hashes and sizes; its independent verification +certificate covers every indexed file. No raw performance journal is committed. + +[Previous receipt-pressure diagnostic](pr67-receipt-admission-measurement.md), +[bundle implementation](bundle-coverage-implementation.md), +[runtime design](../crates/cellule-runtime/docs/write-performance-design.md). From 8bde901c7b5237c5f28f1da70231f1283f0f1b0f Mon Sep 17 00:00:00 2001 From: forhappy Date: Thu, 8 Oct 2026 23:00:10 -0700 Subject: [PATCH 059/102] Observe native submission waits before follower issuance --- .../cellule-axum/examples/sql_metrics/mod.rs | 11 ++- .../examples/sql_metrics/submission.rs | 69 ++++++++++++++ crates/cellule-runtime/docs/runtime.md | 12 +++ crates/cellule-runtime/src/fleet/telemetry.rs | 47 ++++++++++ .../src/node/log_shipper/mod.rs | 28 ++++++ .../src/node/log_shipper/submission.rs | 89 +++++++++++++++++++ .../src/node/log_shipper/tests.rs | 36 +++++++- 7 files changed, 289 insertions(+), 3 deletions(-) create mode 100644 crates/cellule-axum/examples/sql_metrics/submission.rs create mode 100644 crates/cellule-runtime/src/node/log_shipper/submission.rs diff --git a/crates/cellule-axum/examples/sql_metrics/mod.rs b/crates/cellule-axum/examples/sql_metrics/mod.rs index ca29483f..9bde586d 100644 --- a/crates/cellule-axum/examples/sql_metrics/mod.rs +++ b/crates/cellule-axum/examples/sql_metrics/mod.rs @@ -1,7 +1,7 @@ //! Finite, nonblocking runtime and provider measurements for this application. use cellule_runtime::fleet::telemetry::{ - CellTelemetry, CommandResponseSource, DurabilitySubmissionOutcome, PrimitiveOperationKind, - PrimitiveOperationOutcome, PublicationTiming, SharedPublicationTiming, + CellTelemetry, CommandResponseSource, DurabilitySubmissionOutcome, NodeLogSubmissionTiming, + PrimitiveOperationKind, PrimitiveOperationOutcome, PublicationTiming, SharedPublicationTiming, }; use cellule_store::{StorageObservation, StorageObserver, StorageOperation, StorageOutcome}; use std::{ @@ -11,6 +11,7 @@ use std::{ mod capture; mod storage; +mod submission; use capture::CaptureMetrics; #[cfg(test)] @@ -37,6 +38,7 @@ pub(super) struct QueryMetrics { log_append_successes: AtomicU64, log_append_failures: AtomicU64, log_append_bytes: AtomicU64, + submission: submission::SubmissionMetrics, selected_roots: AtomicU64, materialized_commits: AtomicU64, shared: SharedMetrics, @@ -288,6 +290,9 @@ impl CellTelemetry for QueryMetrics { } self.log_append_bytes.fetch_add(bytes, Ordering::Relaxed); } + fn node_log_submission(&self, _cell: cellule_runtime::CellId, timing: NodeLogSubmissionTiming) { + self.submission.observe(timing); + } fn command_execution(&self, queue: Duration, worker: Duration, _succeeded: bool) { self.writes.queue.observe(queue); self.writes.worker.observe(worker); @@ -421,6 +426,7 @@ impl QueryMetrics { histograms.insert(name.into(), histogram.raw()); } histograms.extend(self.capture.window_snapshot()); + histograms.extend(self.submission.window_snapshot()); value["histograms"] = histograms.into(); value } @@ -512,6 +518,7 @@ impl QueryMetrics { "failures": self.log_append_failures.load(Ordering::Relaxed), "bytes": self.log_append_bytes.load(Ordering::Relaxed) }, + "node_log_submission": self.submission.snapshot(), "follower_append": { "input_frames": self.follower_frames.load(Ordering::Relaxed), "data_sync_calls": self.follower_sync_calls.load(Ordering::Relaxed), diff --git a/crates/cellule-axum/examples/sql_metrics/submission.rs b/crates/cellule-axum/examples/sql_metrics/submission.rs new file mode 100644 index 00000000..f257e8f5 --- /dev/null +++ b/crates/cellule-axum/examples/sql_metrics/submission.rs @@ -0,0 +1,69 @@ +//! Fixed phases for the same successfully assigned native capture cohort. +use super::*; + +const LABELS: [&str; 8] = [ + "submission_validation", + "submission_native_bytes", + "submission_shipping_slot", + "submission_local_load", + "submission_ordered_lane", + "submission_publication_slot", + "submission_assignment", + "submission_total", +]; + +#[derive(Default)] +pub(super) struct SubmissionMetrics { + successes: AtomicU64, + failures: AtomicU64, + cancelled: AtomicU64, + frames: AtomicU64, + bytes: AtomicU64, + phases: [Histogram; LABELS.len()], +} + +impl SubmissionMetrics { + pub(super) fn observe(&self, timing: NodeLogSubmissionTiming) { + if timing.cancelled { + self.cancelled.fetch_add(1, Ordering::Relaxed); + } else if !timing.succeeded { + self.failures.fetch_add(1, Ordering::Relaxed); + } else { + self.successes.fetch_add(1, Ordering::Relaxed); + self.frames.fetch_add(timing.frames, Ordering::Relaxed); + self.bytes.fetch_add(timing.bytes, Ordering::Relaxed); + // Every phase, including zero waits, describes this same complete + // assignment. Failed/cancelled submissions have separate counters. + for (histogram, duration) in self.phases.iter().zip([ + timing.validation, + timing.native_bytes, + timing.shipping_slot, + timing.local_load, + timing.ordered_lane, + timing.publication_slot, + timing.assignment, + timing.total, + ]) { + histogram.observe(duration); + } + } + } + + pub(super) fn snapshot(&self) -> serde_json::Value { + serde_json::json!({ + "successes": self.successes.load(Ordering::Relaxed), + "failures": self.failures.load(Ordering::Relaxed), + "cancelled": self.cancelled.load(Ordering::Relaxed), + "assigned_frames": self.frames.load(Ordering::Relaxed), + "assigned_bytes": self.bytes.load(Ordering::Relaxed), + }) + } + + pub(super) fn window_snapshot(&self) -> serde_json::Map { + LABELS + .iter() + .zip(&self.phases) + .map(|(label, histogram)| ((*label).into(), histogram.raw())) + .collect() + } +} diff --git a/crates/cellule-runtime/docs/runtime.md b/crates/cellule-runtime/docs/runtime.md index 5ed43424..5c027718 100644 --- a/crates/cellule-runtime/docs/runtime.md +++ b/crates/cellule-runtime/docs/runtime.md @@ -305,6 +305,18 @@ read-only setup, and handler execution. Compare it with the registered primitive's duration to distinguish SQL work from dispatch and waiting. Pre-dispatch refusals are excluded; SQL deadline failures are included. +`CellTelemetry::node_log_submission` partitions one capture's lifetime after SQL +and before its follower-log assignment: validation, native-byte admission, +shipping slot, local load/validation, ordered lane, publication slot and final +assignment. The same observation carries the original logical range and reports +completion, failure or cancellation. A slow publication feed can block issuance +while the ordered mutex is held; this wait is outside worker and proof timers. +The callback grants no assignment or durability. Sinks must remain nonblocking +and use finite phase labels, never Cell identities or sequences as labels. +The SQL example exports matching successful-assignment histograms and separate +failure/cancellation counts. Compare phases from that same cohort; response and +background-root histograms describe different lifetimes. + **Automatic rollback** - SQLite may automatically roll back the whole command on capacity or interruption errors. The managed LTX writer recognizes completed rollback using autocommit and its WAL commit observer, preserves the original error, and keeps the Cell servable. diff --git a/crates/cellule-runtime/src/fleet/telemetry.rs b/crates/cellule-runtime/src/fleet/telemetry.rs index 126ab7f1..f35d788a 100644 --- a/crates/cellule-runtime/src/fleet/telemetry.rs +++ b/crates/cellule-runtime/src/fleet/telemetry.rs @@ -73,6 +73,43 @@ pub struct FollowerAppendTiming { pub succeeded: bool, } +/// One native capture's submission before follower durability can be observed. +/// +/// The durations partition the same submission lifetime, including cancelled +/// futures. They exclude SQL execution, subsequent shipping and object selection. +/// Cell identity and sequences are for local trace correlation, never labels. +#[derive(Clone, Copy, Debug, Default)] +pub struct NodeLogSubmissionTiming { + /// Original inclusive logical command range represented by this capture. + pub first_commit_sequence: u64, + /// Original inclusive logical command range endpoint. + pub commit_sequence: u64, + /// Number of native frames in the submitted capture. + pub frames: u64, + /// Expected canonical framed bytes admitted by the native lane. + pub bytes: u64, + /// Initial scope, length and capacity validation. + pub validation: Duration, + /// Waiting for the original outstanding native-byte reservation. + pub native_bytes: Duration, + /// Acquiring the shipping sender and reserving its bounded queue slot. + pub shipping_slot: Duration, + /// Blocking-worker dispatch, local file reads and native frame validation. + pub local_load: Duration, + /// Waiting for the original ordered issuance mutex. + pub ordered_lane: Duration, + /// Waiting for the bounded publication feed while owning the issuance mutex. + pub publication_slot: Duration, + /// Ticket validation, final encoding, assignment and enqueueing both consumers. + pub assignment: Duration, + /// Entire submission lifetime, partitioned by the phases above. + pub total: Duration, + /// Whether one complete original capture was assigned and enqueued. + pub succeeded: bool, + /// Whether the submission future was dropped before returning a result. + pub cancelled: bool, +} + /// Outcome of an actor-owned resident route lookup. #[derive(Clone, Copy, Debug, PartialEq, Eq)] pub enum ResidentRouteOutcome { @@ -256,6 +293,10 @@ pub trait CellTelemetry: Send + Sync { /// Records bytes sent to follower append lanes and whether every lane acknowledged them. fn node_log_append(&self, _acknowledged: bool, _bytes: u64) {} + /// Records pre-issuance waits for one complete native capture. This grants + /// no ticket or durability proof. Cancellation is observed without new work. + fn node_log_submission(&self, _cell: CellId, _timing: NodeLogSubmissionTiming) {} + /// Reports native follower work separately from peer HTTP and authorization. fn follower_append(&self, _timing: FollowerAppendTiming) {} @@ -441,6 +482,12 @@ impl CellTelemetryHandle { } } + pub(crate) fn node_log_submission(&self, cell: CellId, timing: NodeLogSubmissionTiming) { + if let Some(telemetry) = self.inner.get() { + telemetry.node_log_submission(cell, timing); + } + } + pub(crate) fn follower_append(&self, timing: FollowerAppendTiming) { if let Some(telemetry) = self.inner.get() { telemetry.follower_append(timing); diff --git a/crates/cellule-runtime/src/node/log_shipper/mod.rs b/crates/cellule-runtime/src/node/log_shipper/mod.rs index 93af51dc..a0d86fb5 100644 --- a/crates/cellule-runtime/src/node/log_shipper/mod.rs +++ b/crates/cellule-runtime/src/node/log_shipper/mod.rs @@ -12,6 +12,7 @@ use crate::node::log_transport::{AppendRequest, NodeLogTransport}; use crate::{Error, Result}; mod publication; +mod submission; pub use publication::{AssignedCapture, NodePublicationFeed, SelectedBundlePublication}; pub(crate) use publication::{SelectedBundle, SubmittedCapture}; @@ -201,6 +202,7 @@ pub struct NodeLogShipper { limits: cellule_ltx::Limits, publication: publication::PublicationState, stopping: tokio::sync::watch::Sender, + telemetry: crate::fleet::telemetry::CellTelemetryHandle, } impl NodeLogShipper { @@ -266,6 +268,7 @@ impl NodeLogShipper { limits, publication: publication::PublicationState::default(), stopping, + telemetry, }) } @@ -322,7 +325,26 @@ impl NodeLogShipper { &self, submission: NodeLogSubmission, ) -> Result { + let mut observation = submission::Observation::new( + self.telemetry.clone(), + submission.cell, + submission.first_commit_sequence, + submission.commit_sequence, + submission.encoded_bytes, + ); + let result = self.assign_capture(submission, &mut observation).await; + observation.finish(result.is_ok()); + result + } + + async fn assign_capture( + &self, + submission: NodeLogSubmission, + observation: &mut submission::Observation, + ) -> Result { + use submission::Stage; let frame_count = submission.frame_count()?; + observation.frames(frame_count); if submission .segments .iter() @@ -336,10 +358,12 @@ impl NodeLogShipper { .ok() .filter(|bytes| *bytes != 0) .ok_or(Error::Capacity("node-log outstanding bytes"))?; + observation.enter(Stage::NativeBytes); let reservation = Arc::clone(&self.bytes) .acquire_many_owned(permit_count) .await .map_err(|_| Error::RuntimeClosed)?; + observation.enter(Stage::ShippingSlot); let sender = self .sender .lock() @@ -350,6 +374,7 @@ impl NodeLogShipper { .reserve_owned() .await .map_err(|_| Error::RuntimeClosed)?; + observation.enter(Stage::LocalLoad); let limits = self.limits; let (leader, log_epoch, _) = self.gate.shipping_scope()?; let loaded = @@ -359,13 +384,16 @@ impl NodeLogShipper { // Expensive LTX validation is parallel and bounded by outstanding-byte // admission. This lane only patches exclusively owned envelopes and // atomically commits their consecutive ticket before enqueueing. + observation.enter(Stage::OrderedLane); let _ordered = self.order.lock().await; // Reserve both consumers before committing a sequence. Cancellation or // a full publication queue therefore cannot leave an unselectable gap. + observation.enter(Stage::PublicationSlot); let publication = self.publication.reserve(&self.stopping).await?; if *self.stopping.borrow() { return Err(Error::RuntimeClosed); } + observation.enter(Stage::Assignment); let ticket = self.gate.preview(frame_count)?; let encoded = loaded.encode(ticket)?; let assignment = self diff --git a/crates/cellule-runtime/src/node/log_shipper/submission.rs b/crates/cellule-runtime/src/node/log_shipper/submission.rs new file mode 100644 index 00000000..3b848648 --- /dev/null +++ b/crates/cellule-runtime/src/node/log_shipper/submission.rs @@ -0,0 +1,89 @@ +//! One partitioned submission lifetime, including cancellation during admission. +use std::time::Instant; + +use crate::CellId; +use crate::fleet::telemetry::{CellTelemetryHandle, NodeLogSubmissionTiming}; + +pub(super) enum Stage { + Validation, + NativeBytes, + ShippingSlot, + LocalLoad, + OrderedLane, + PublicationSlot, + Assignment, +} + +pub(super) struct Observation { + telemetry: CellTelemetryHandle, + cell: CellId, + started: Instant, + phase_started: Instant, + stage: Stage, + timing: NodeLogSubmissionTiming, +} + +impl Observation { + pub(super) fn new( + telemetry: CellTelemetryHandle, + cell: CellId, + first_commit_sequence: u64, + commit_sequence: u64, + bytes: u64, + ) -> Self { + let started = Instant::now(); + Self { + telemetry, + cell, + started, + phase_started: started, + stage: Stage::Validation, + timing: NodeLogSubmissionTiming { + first_commit_sequence, + commit_sequence, + bytes, + cancelled: true, + ..NodeLogSubmissionTiming::default() + }, + } + } + + pub(super) fn frames(&mut self, frames: u64) { + self.timing.frames = frames; + } + + pub(super) fn enter(&mut self, stage: Stage) { + let now = Instant::now(); + self.observe_phase(now); + self.phase_started = now; + self.stage = stage; + } + + pub(super) fn finish(&mut self, succeeded: bool) { + self.timing.succeeded = succeeded; + self.timing.cancelled = false; + } + + fn observe_phase(&mut self, now: Instant) { + let duration = now.saturating_duration_since(self.phase_started); + let phase = match self.stage { + Stage::Validation => &mut self.timing.validation, + Stage::NativeBytes => &mut self.timing.native_bytes, + Stage::ShippingSlot => &mut self.timing.shipping_slot, + Stage::LocalLoad => &mut self.timing.local_load, + Stage::OrderedLane => &mut self.timing.ordered_lane, + Stage::PublicationSlot => &mut self.timing.publication_slot, + Stage::Assignment => &mut self.timing.assignment, + }; + *phase += duration; + } +} + +impl Drop for Observation { + fn drop(&mut self) { + let now = Instant::now(); + self.observe_phase(now); + self.timing.total = now.saturating_duration_since(self.started); + self.telemetry.node_log_submission(self.cell, self.timing); + } +} diff --git a/crates/cellule-runtime/src/node/log_shipper/tests.rs b/crates/cellule-runtime/src/node/log_shipper/tests.rs index a6c26aa5..9d632242 100644 --- a/crates/cellule-runtime/src/node/log_shipper/tests.rs +++ b/crates/cellule-runtime/src/node/log_shipper/tests.rs @@ -19,12 +19,20 @@ struct RecordingTransport { #[derive(Default)] struct RecordingTelemetry { appends: Mutex>, + submissions: Mutex>, } impl crate::fleet::telemetry::CellTelemetry for RecordingTelemetry { fn node_log_append(&self, acknowledged: bool, bytes: u64) { self.appends.lock().unwrap().push((acknowledged, bytes)); } + fn node_log_submission( + &self, + cell: CellId, + timing: crate::fleet::telemetry::NodeLogSubmissionTiming, + ) { + self.submissions.lock().unwrap().push((cell, timing)); + } } struct LostAckTransport { @@ -388,10 +396,12 @@ async fn cancelled_full_publication_queue_does_not_issue_a_native_gap() { let (_directory, cuts) = capture(); let gate = DurabilityGate::new(session(1), node(1), 2, [node(2)]).unwrap(); gate.activate_fleet().unwrap(); - let shipper = NodeLogShipper::new( + let telemetry = Arc::new(RecordingTelemetry::default()); + let shipper = NodeLogShipper::new_with_telemetry( gate.clone(), Arc::new(RecordingTransport::default()), cellule_ltx::Limits::default(), + crate::fleet::telemetry::CellTelemetryHandle::from_sink(telemetry.clone()), ) .unwrap(); let mut feed = shipper.take_publication_feed().unwrap(); @@ -410,6 +420,30 @@ async fn cancelled_full_publication_queue_does_not_issue_a_native_gap() { .is_err() ); assert_eq!(gate.issued_through(), 512); + { + let observed = telemetry.submissions.lock().unwrap(); + assert_eq!(observed.len(), 513); + let (cell, blocked) = observed.last().unwrap(); + assert_eq!(*cell, publication_submission(&cuts, 512).cell); + assert!(blocked.cancelled); + assert!(!blocked.succeeded); + assert!(blocked.publication_slot > Duration::ZERO); + for (_, timing) in observed.iter() { + assert_eq!( + timing.validation + + timing.native_bytes + + timing.shipping_slot + + timing.local_load + + timing.ordered_lane + + timing.publication_slot + + timing.assignment, + timing.total + ); + assert_eq!(timing.first_commit_sequence, 4); + assert_eq!(timing.commit_sequence, 4); + assert_eq!(timing.frames, 1); + } + } let first = feed.recv().await.unwrap(); assert_eq!(first.assignment().ticket().first_sequence(), 1); first.assignment().verify(first.frames()).unwrap(); From e7b93962f2ebfd084692a1eccb0475cd1355f4fb Mon Sep 17 00:00:00 2001 From: forhappy Date: Thu, 8 Oct 2026 23:15:09 -0700 Subject: [PATCH 060/102] Record publication wait behind global issuance lock --- .../docs/write-performance-design.md | 6 + docs/bundle-coverage-implementation.md | 6 + docs/pr67-submission-timing-measurement.md | 110 ++++++++++++++++++ 3 files changed, 122 insertions(+) create mode 100644 docs/pr67-submission-timing-measurement.md diff --git a/crates/cellule-runtime/docs/write-performance-design.md b/crates/cellule-runtime/docs/write-performance-design.md index de37ae5f..7e0ba284 100644 --- a/crates/cellule-runtime/docs/write-performance-design.md +++ b/crates/cellule-runtime/docs/write-performance-design.md @@ -1,5 +1,11 @@ # Node write and read performance design +The [submission-timing diagnostic](../../../docs/pr67-submission-timing-measurement.md) +now measures the missing pre-proof queue: 727.30 ms waiting for the global +issuance lock, whose holder waits 6.14 ms for publication capacity. The +instrumented application completes 158.37 writes/s, with all 19,675 ACKs passing +warm/cold audit. This is diagnosis, not a throughput improvement or qualification. + The [latest replay-pressure verification](../../../docs/pr67-replay-admission-measurement.md) preserves original durable outcomes under publication pressure. One fresh Fleet pair completes 158.15 writes/s versus 152.92 before, with all 19,542 candidate ACKs diff --git a/docs/bundle-coverage-implementation.md b/docs/bundle-coverage-implementation.md index 30dbab0f..7c747e5c 100644 --- a/docs/bundle-coverage-implementation.md +++ b/docs/bundle-coverage-implementation.md @@ -1,5 +1,11 @@ # Bundle coverage implementation +The [submission-timing diagnostic](pr67-submission-timing-measurement.md) +attributes the main pre-proof delay to publication capacity held under the global +issuance lock: 727.30 ms mean lock wait, with a 6.14-ms publication wait inside +the lock. Its 158.37 writes/s result is unqualified; all 19,675 ACKs pass warm/cold +audit. No throughput gain is claimed for this instrumentation slice. + The [latest replay-pressure diagnostic](pr67-replay-admission-measurement.md) preserves original durable command outcomes under node and Cell publication pressure. One fresh Fleet pair completes 158.15 writes/s versus 152.92 before diff --git a/docs/pr67-submission-timing-measurement.md b/docs/pr67-submission-timing-measurement.md new file mode 100644 index 00000000..272de408 --- /dev/null +++ b/docs/pr67-submission-timing-measurement.md @@ -0,0 +1,110 @@ +# PR 67: publication stalls the global issuance lane + +**The measured bottleneck is a publication wait inside the global issuance +lock.** Instrumented revision `8bde901c7b5237c5f28f1da70231f1283f0f1b0f` +completes 158.37 Fleet writes/s. Across 9,719 successful native submissions, +mean submission time is 733.65 ms: 727.30 ms waiting for the ordered lock and +6.14 ms waiting for publication capacity while holding it. This is diagnosis, +not a delivered throughput improvement. PR #67 remains a draft. + +## Measurement and limits + +The unchanged SQL-ledger diagnostic uses 1,000 uniform Cells, 96-byte values, +128 clients/queue slots, 15K offered writes/s, 30-second warmup and a 60-second +window. Retained credit is 1 GiB; managed disk remains 1 GiB. Client, auditor, +fixtures and images match the previous large-credit control. No build or test +suite overlaps the window. Managed SQLite already uses WAL NORMAL. + +| Result | Observation | +| --- | ---: | +| Successful in-window writes | 9,502; 158.37/s | +| Successful trailing writes | 256 | +| Request errors | 0 | +| Dropped measured offers | 890,242 | +| Successful scheduled p99 | 3,951.43 ms | +| Successful request p99 | 2,401.45 ms | +| Complete ACK cohort | 19,675 | +| Warm/cold reads and original-outcome retries | All 19,675 pass in each audit; zero errors | +| Joined original Fleet drain | 31.44 s | + +Every offer, attempt, completion and complete ACK count reconciles independently. +Provider health and lifecycle checks pass. Performance qualification fails. +The prior same-profile candidate observation was 191.08/s: this new unpaired run +is 17.12% lower, with worse tail latency. It establishes neither an improvement +nor a repeatable regression attributable to instrumentation. + +The shared Docker VM has 8 CPUs and 8 GiB total RAM; container ceilings exceed +that capacity. This is not dedicated 8-vCPU/16-GiB standard-node qualification. +There is no fresh paired celld, Bucket or read-only result in this slice. +The previous celld observation, 4,473.32/s, has asymmetric internal resource +policies, dropped offers and failed warm recovery; it is not qualified capacity. + +## Exact submission partition + +All eight histograms have 9,719 observations, matching successful assignment +callbacks. Their summed phase durations equal the total exactly; no failed or +cancelled submissions are observed in this window. This cohort includes +assignment completions across the window boundaries and differs from the +9,502 in-window HTTP responses. It excludes SQL, later follower proof and reply. + +| Submission phase | Mean ms | +| --- | ---: | +| Validation | 0.00014 | +| Native byte credit | 0.00025 | +| Shipping slot | 0.00029 | +| Local load and validation | 0.19809 | +| Ordered issuance lock | 727.30445 | +| Publication slot, with lock held | 6.14275 | +| Ticket assignment and enqueue | 0.00702 | +| Total | 733.65298 | + +The ordered-lock wait accounts for 99.13% of submission time. Publication-slot +waits sum to 59.70 seconds in this completion cohort. A serial 6.15-ms +publication/assignment interval permits approximately 163 submissions/s, +consistent with the observed 158.37 HTTP completions/s. This is a queueing +interpretation, not a prediction of capacity after a future change. + +[`assign_capture`](../crates/cellule-runtime/src/node/log_shipper/mod.rs) +holds the ordered mutex across publication-slot reservation, then issues the +ticket and enqueues both consumers. A slow selector therefore queues every +writer before follower proof starts. The queue deliberately prevents sequence +gaps and bounds original publication debt; removing those contracts is not a fix. +Moving the same capacity wait outside the mutex alone would move the measured +wait without increasing the selector's sustainable consumption rate. + +Background cost remains high: 14.40 GET/range attempts and 0.618 successful PUTs +per in-window completion; 11.43 materialized commands per root. These API counters +include background cohorts and exclude SDK retries, so they are not exact +per-command costs. Historical ranges and base dependencies are checked again +for each affected binding. Mean worker and Fleet-proof timers are 2.05 and +7.40 ms in their respective overlapping cohorts; they are not additive to this +submission partition. + +Celld's [Fleet loop](https://github.com/denoland/celld/blob/f2bf648663a610eefde71f3547ad61e9b896b1f0/crates/celld/ltx_repl.rs#L4514) +explicitly avoids bucket waits and pipelines ordered rounds. Its +[follower stream](https://github.com/denoland/celld/blob/f2bf648663a610eefde71f3547ad61e9b896b1f0/crates/celld/node_log.rs#L160) +groups already delivered appends before fsync. Sharing SQLite, LTX and follower +logs does not imply the same scheduling, proof work or acknowledgement path. + +The next implementation must lower selection I/O and safely separate original +native progress from publication debt, with bounded recoverable backlog, +complete issued-suffix recovery and joined drain. Pipeline follower rounds after +addressing this dominant pre-proof queue. Qualification targets stay unchanged. + +## Verification and evidence + +The instrumentation preserves admission, ticket order, byte limits and durability +policy. Cancellation telemetry is exercised by the real full-publication-queue +regression, including no sequence gap and zero retained credit. All contributor +routes pass in an immutable snapshot: 1,965 workspace tests, 60 local LTX tests, +both Rust 1.97 and 1.99 Clippy, docs, layout, boundaries and document/peer gates. +The 38 environment-dependent ignored tests remain unqualified. + +Raw source, binaries, fixtures, verification, metrics, journals and independent +reconciliation stay outside Git under +`/Volumes/Workspace/crabbuild-target/cellule-write-perf-8ad1/submission-timing-20261008-*`. +The external index and rehash certificate record the complete evidence inventory. + +[Previous replay-pressure comparison](pr67-replay-admission-measurement.md), +[implementation](bundle-coverage-implementation.md), +[capacity contract](../crates/cellule-runtime/docs/write-performance-design.md). From 90f099a295158c6effe51079d35db07eab204d17 Mon Sep 17 00:00:00 2001 From: forhappy Date: Thu, 8 Oct 2026 23:31:20 -0700 Subject: [PATCH 061/102] Coalesce fresh historical ranges within original memory admission --- crates/cellule-runtime/src/node/bundle/mod.rs | 1 + .../cellule-runtime/src/node/bundle/origin.rs | 7 +- .../cellule-runtime/src/node/bundle/proof.rs | 144 +++++--- .../src/node/bundle/selection.rs | 36 +- .../src/node/bundle/tests/faults.rs | 9 + .../node/bundle/tests/index/history_cohort.rs | 309 ++++++++++++++++++ .../src/node/bundle/tests/index/mod.rs | 1 + .../src/node/bundle/verification/mod.rs | 262 +++++++++++++++ .../src/node/bundle/verification/tests.rs | 105 ++++++ .../src/node/durability/publication/mod.rs | 5 +- 10 files changed, 815 insertions(+), 64 deletions(-) create mode 100644 crates/cellule-runtime/src/node/bundle/tests/index/history_cohort.rs create mode 100644 crates/cellule-runtime/src/node/bundle/verification/mod.rs create mode 100644 crates/cellule-runtime/src/node/bundle/verification/tests.rs diff --git a/crates/cellule-runtime/src/node/bundle/mod.rs b/crates/cellule-runtime/src/node/bundle/mod.rs index 978c7ec6..b475d889 100644 --- a/crates/cellule-runtime/src/node/bundle/mod.rs +++ b/crates/cellule-runtime/src/node/bundle/mod.rs @@ -46,6 +46,7 @@ mod origin; mod proof; pub(crate) mod recovery; mod selection; +mod verification; pub(crate) use continuation::MaterializedBundlePrefix; #[cfg(test)] use proof::checkpoint_prefix; diff --git a/crates/cellule-runtime/src/node/bundle/origin.rs b/crates/cellule-runtime/src/node/bundle/origin.rs index 79b87a8a..3bae012d 100644 --- a/crates/cellule-runtime/src/node/bundle/origin.rs +++ b/crates/cellule-runtime/src/node/bundle/origin.rs @@ -31,11 +31,14 @@ impl OriginBundle { Ok(Self { session: prepared.catalog.session, head: prepared.head, - body, + // Reuse the proposal's allocation only after this operation's + // complete fresh read matched it. The original 4-MiB read buffer + // can then cover bounded historical scratch and checked facts. + body: prepared.body.clone(), }) } - fn range( + pub(super) fn range( &self, session: SessionId, epoch: u64, diff --git a/crates/cellule-runtime/src/node/bundle/proof.rs b/crates/cellule-runtime/src/node/bundle/proof.rs index c19ca8ba..3852ffd5 100644 --- a/crates/cellule-runtime/src/node/bundle/proof.rs +++ b/crates/cellule-runtime/src/node/bundle/proof.rs @@ -136,8 +136,6 @@ pub(super) async fn verify_selected_binding( limits: cellule_ltx::Limits, origin: &origin::OriginBundle, ) -> Result<()> { - // Selection needs exact locators, not retained native bodies. Keep the same - // verifier as reconstruction while dropping each checked frame promptly. verify_binding_into(layout, session, epoch, binding, limits, Some(origin), None).await } @@ -151,19 +149,7 @@ async fn verify_binding_into( mut frames: Option<&mut Vec>, ) -> Result<()> { verify_base(layout, binding, limits).await?; - let mut position = binding - .control - .ltx_root() - .ok_or(Error::Node("bundle base absent"))? - .position; - let mut commit = binding - .control - .root - .as_ref() - .ok_or(Error::Node("bundle base absent"))? - .commit_sequence; - let mut sequence = 0; - let mut first_commit = commit; + let mut chain = BindingChain::new(binding)?; for locator in &binding.locators { let object = locator .object @@ -174,43 +160,113 @@ async fn verify_binding_into( .ok_or(Error::Node("bundle locator overflow"))?; let bytes = origin::read_range(layout, session, epoch, object, locator.offset..end, origin).await?; - if bytes.len() as u64 != locator.bytes - || *blake3::hash(&bytes).as_bytes() != *locator.frame_digest.as_bytes() - { - return Err(Error::Node("bundle frame digest differs")); + let frame = checked_frame(session, epoch, binding, locator, bytes, limits)?; + chain.accept(FrameStep::from_frame(&frame))?; + if let Some(frames) = &mut frames { + frames.push(frame); } - let frame = cellule_ltx::inspect_node_frame(bytes, limits)?; - let scope = frame.scope(); - if scope.leader_session != *session.as_bytes() - || scope.log_epoch != epoch - || scope.application != *binding.application.as_bytes() - || scope.cell != *binding.control.cell.as_bytes() - || scope.incarnation != *binding.control.incarnation.as_bytes() - || scope.cell_epoch != binding.control.epoch - || scope.node_sequence <= sequence - || (scope.commit_sequence == commit && frame.first_commit_sequence() != first_commit) - || (scope.commit_sequence != commit - && commit.checked_add(1) != Some(frame.first_commit_sequence())) - || position.txid.checked_add(1) != Some(frame.segment().min_txid) - || position.checksum != frame.segment().pre_checksum + } + chain.finish(binding) +} + +/// Checked frame facts retained only within one verification operation. They +/// grant no proof or availability outside that operation and hold no body. +#[derive(Clone, Copy)] +pub(super) struct FrameStep { + sequence: u64, + first_commit: u64, + commit: u64, + min_txid: u64, + pre_checksum: u64, + position: cellule_ltx::Position, +} + +impl FrameStep { + pub(super) fn from_frame(frame: &cellule_ltx::VerifiedNodeFrame) -> Self { + Self { + sequence: frame.scope().node_sequence, + first_commit: frame.first_commit_sequence(), + commit: frame.scope().commit_sequence, + min_txid: frame.segment().min_txid, + pre_checksum: frame.segment().pre_checksum, + position: frame.segment().position(), + } + } +} + +pub(super) struct BindingChain { + position: cellule_ltx::Position, + commit: u64, + sequence: u64, + first_commit: u64, +} + +impl BindingChain { + pub(super) fn new(binding: &Binding) -> Result { + let base = binding + .control + .ltx_root() + .ok_or(Error::Node("bundle base absent"))?; + Ok(Self { + position: base.position, + commit: base.commit_sequence, + sequence: 0, + first_commit: base.commit_sequence, + }) + } + + pub(super) fn accept(&mut self, step: FrameStep) -> Result<()> { + if step.sequence <= self.sequence + || (step.commit == self.commit && step.first_commit != self.first_commit) + || (step.commit != self.commit && self.commit.checked_add(1) != Some(step.first_commit)) + || self.position.txid.checked_add(1) != Some(step.min_txid) + || self.position.checksum != step.pre_checksum { return Err(Error::Node("bundle locator violates exact Cell range")); } - first_commit = frame.first_commit_sequence(); - position = frame.segment().position(); - commit = scope.commit_sequence; - sequence = scope.node_sequence; - if let Some(frames) = &mut frames { - frames.push(frame); + self.position = step.position; + self.commit = step.commit; + self.sequence = step.sequence; + self.first_commit = step.first_commit; + Ok(()) + } + + pub(super) fn finish(self, binding: &Binding) -> Result<()> { + if self.position != binding.selected_position + || self.commit != binding.selected_commit + || (!binding.locators.is_empty() && self.sequence != binding.selected_sequence) + { + return Err(Error::Node("bundle proof endpoint differs")); } + Ok(()) + } +} + +pub(super) fn checked_frame( + session: SessionId, + epoch: u64, + binding: &Binding, + locator: &Locator, + bytes: Bytes, + limits: cellule_ltx::Limits, +) -> Result { + if bytes.len() as u64 != locator.bytes + || *blake3::hash(&bytes).as_bytes() != *locator.frame_digest.as_bytes() + { + return Err(Error::Node("bundle frame digest differs")); } - if position != binding.selected_position - || commit != binding.selected_commit - || (!binding.locators.is_empty() && sequence != binding.selected_sequence) + let frame = cellule_ltx::inspect_node_frame(bytes, limits)?; + let scope = frame.scope(); + if scope.leader_session != *session.as_bytes() + || scope.log_epoch != epoch + || scope.application != *binding.application.as_bytes() + || scope.cell != *binding.control.cell.as_bytes() + || scope.incarnation != *binding.control.incarnation.as_bytes() + || scope.cell_epoch != binding.control.epoch { - return Err(Error::Node("bundle proof endpoint differs")); + return Err(Error::Node("bundle locator violates exact Cell range")); } - Ok(()) + Ok(frame) } pub(super) async fn verify_base( diff --git a/crates/cellule-runtime/src/node/bundle/selection.rs b/crates/cellule-runtime/src/node/bundle/selection.rs index 04c80da5..e273e44e 100644 --- a/crates/cellule-runtime/src/node/bundle/selection.rs +++ b/crates/cellule-runtime/src/node/bundle/selection.rs @@ -241,23 +241,27 @@ impl NodeDirectory { Some(&origin), ) .await?; + let bindings = catalog + .bindings + .into_iter() + .filter(|binding| { + cells.contains(&( + *binding.application.as_bytes(), + *binding.control.cell.as_bytes(), + )) + }) + .collect::>(); + verification::verify_cohort( + &self.layout, + catalog.session, + catalog.epoch, + &bindings, + limits, + &origin, + ) + .await?; let mut proofs = Vec::new(); - for binding in catalog.bindings { - if !cells.contains(&( - *binding.application.as_bytes(), - *binding.control.cell.as_bytes(), - )) { - continue; - } - proof::verify_selected_binding( - &self.layout, - catalog.session, - catalog.epoch, - &binding, - limits, - &origin, - ) - .await?; + for binding in bindings { let pin = binding .control .bundle_binding diff --git a/crates/cellule-runtime/src/node/bundle/tests/faults.rs b/crates/cellule-runtime/src/node/bundle/tests/faults.rs index 66b2d2c2..4233cf1d 100644 --- a/crates/cellule-runtime/src/node/bundle/tests/faults.rs +++ b/crates/cellule-runtime/src/node/bundle/tests/faults.rs @@ -12,6 +12,7 @@ pub(super) struct ReplyFault { pub(super) mode: AtomicU8, pub(super) node_updates: AtomicUsize, pub(super) coverage_puts: AtomicUsize, + pub(super) range_started: AtomicUsize, pub(super) pin_started: tokio::sync::Notify, pub(super) pin_resume: tokio::sync::Notify, pub(super) node_started: tokio::sync::Notify, @@ -117,6 +118,14 @@ impl ObjectStore for ReplyFault { self.inner.put_multipart_opts(path, opts).await } async fn get_opts(&self, path: &Path, opts: GetOptions) -> object_store::Result { + if self.mode.load(Ordering::SeqCst) == 11 + && path.as_ref().ends_with(".cnb") + && opts.range.is_some() + { + self.range_started.fetch_add(1, Ordering::SeqCst); + self.node_started.notify_one(); + self.node_resume.notified().await; + } self.inner.get_opts(path, opts).await } fn delete_stream( diff --git a/crates/cellule-runtime/src/node/bundle/tests/index/history_cohort.rs b/crates/cellule-runtime/src/node/bundle/tests/index/history_cohort.rs new file mode 100644 index 00000000..1ac706b2 --- /dev/null +++ b/crates/cellule-runtime/src/node/bundle/tests/index/history_cohort.rs @@ -0,0 +1,309 @@ +use super::*; + +#[tokio::test] +async fn large_historical_extent_keeps_canonical_serial_verification_and_cold_recovery() { + let mut f = Fixture::new().await; + super::super::coverage::enroll(&mut f).await; + let mut cell = f.cell(4).await; + let mut payload = vec![0_u8; (MAX_BUNDLE_BYTES / 2 + 128 * 1024) as usize]; + let mut hash = blake3::Hasher::new(); + hash.update(b"large-historical-extent"); + hash.finalize_xof().fill(&mut payload); + cell.db + .transaction(|tx| { + tx.execute_batch("CREATE TABLE payload(value BLOB)")?; + tx.execute("INSERT INTO payload VALUES (?1)", [&payload])?; + Ok(()) + }) + .unwrap(); + let (_, frames, assigned) = f.append(&mut cell, 2); + assert!( + frames + .iter() + .any(|frame| frame.encoded().len() as u64 > MAX_BUNDLE_BYTES / 2) + ); + let first = f + .directory + .prepare_node_bundle(&f.node, &frames, &[assigned], NOW) + .await + .unwrap(); + f.node = f + .directory + .select_node_bundle(&f.node, &first, &f.lease, Limits::default(), NOW) + .await + .unwrap() + .0; + let (_, frames, assigned) = f.append(&mut cell, 3); + let next = f + .directory + .prepare_node_bundle(&f.node, &frames, &[assigned], NOW) + .await + .unwrap(); + let (_, proofs) = f + .directory + .select_node_bundle(&f.node, &next, &f.lease, Limits::default(), NOW) + .await + .unwrap(); + let proof = &proofs[0]; + assert_eq!(proof.commit_sequence(), 3); + let overlay = proof + .recovery_overlay(&f.layout, Limits::default()) + .await + .unwrap(); + let recovered = cell + .replica + .prepare_recovered_overlay(&overlay, 1) + .await + .unwrap(); + let cold = f.scratch.path().join("large-extent-cold.sqlite"); + cell.replica + .open_root(&recovered.root()) + .await + .unwrap() + .restore(&cold) + .await + .unwrap(); + let db = rusqlite::Connection::open(cold).unwrap(); + let actual: Vec = db + .query_row("SELECT value FROM payload", [], |row| row.get(0)) + .unwrap(); + assert_eq!(actual, payload); + let outcomes: i64 = db + .query_row("SELECT count(*) FROM outcomes", [], |row| row.get(0)) + .unwrap(); + assert_eq!(outcomes, 3); +} + +#[tokio::test] +async fn selection_groups_fresh_historical_extents_across_sixty_four_cells() { + let mut f = Fixture::new().await; + super::super::coverage::enroll(&mut f).await; + let mut cells = Vec::new(); + for number in 4..(4 + MAX_FRAMES as u8) { + cells.push(f.cell(number).await); + f.heartbeat().await; + } + let mut frames = Vec::new(); + let mut assignments = Vec::new(); + for cell in &mut cells { + let (_, capture, assignment) = f.append(cell, 2); + frames.extend(capture); + assignments.push(assignment); + } + let native_bytes = frames + .iter() + .map(|frame| frame.encoded().len()) + .sum::(); + let now = f.node.advertisement().issued_at_ms(); + let first = f + .directory + .prepare_node_bundle(&f.node, &frames, &assignments, now) + .await + .unwrap(); + let (node, _) = f + .directory + .select_node_bundle(&f.node, &first, &f.lease, Limits::default(), now) + .await + .unwrap(); + f.node = node; + let path = f.layout.node_coverage_bundle_path( + f.node.advertisement().session().as_bytes(), + first.head.epoch, + first.head.digest.as_bytes(), + ); + frames.clear(); + assignments.clear(); + for cell in &mut cells { + let (_, capture, assignment) = f.append(cell, 3); + frames.extend(capture); + assignments.push(assignment); + } + let prepared = f + .directory + .prepare_node_bundle(&f.node, &frames, &assignments, now) + .await + .unwrap(); + f.count.reset(); + let (node, proofs) = f + .directory + .select_node_bundle(&f.node, &prepared, &f.lease, Limits::default(), now) + .await + .unwrap(); + let reads = f + .count + .requests() + .into_iter() + .filter(|request| request.location == path.as_ref()) + .collect::>(); + eprintln!( + "historical cohort: cells={} native_bytes={} reads={}", + cells.len(), + native_bytes, + reads.len() + ); + assert_eq!( + reads.len(), + 1, + "one fresh contiguous historical window, rather than a request per Cell" + ); + assert_eq!(proofs.len(), MAX_FRAMES); + for proof in &proofs { + assert_eq!(proof.commit_sequence(), 3); + assert_eq!(proof.locator_count(), 2); + let restored = proof + .recovery_overlay(&f.layout, Limits::default()) + .await + .unwrap(); + assert_eq!(restored.final_commit_sequence(), 3); + let cell = cells + .iter() + .find(|cell| cell.control.value().bundle_binding == Some(proof.binding())) + .unwrap(); + let recovered = cell + .replica + .prepare_recovered_overlay(&restored, 1) + .await + .unwrap(); + let cold = f.scratch.path().join(format!( + "cohort-cold-{}.sqlite", + cell.control.value().cell.as_bytes()[0] + )); + cell.replica + .open_root(&recovered.root()) + .await + .unwrap() + .restore(&cold) + .await + .unwrap(); + let db = rusqlite::Connection::open(cold).unwrap(); + let outcomes: Vec<(String, String)> = db + .prepare("SELECT request, result FROM outcomes ORDER BY request") + .unwrap() + .query_map([], |row| Ok((row.get(0)?, row.get(1)?))) + .unwrap() + .collect::>() + .unwrap(); + assert_eq!( + outcomes, + vec![ + ("request-2".into(), "result-2".into()), + ("request-3".into(), "result-3".into()), + ("seed".into(), "original".into()) + ] + ); + } + f.node = node; + let (body, _) = f + .layout + .store() + .get_with_etag_bounded(&path, MAX_BUNDLE_BYTES) + .await + .unwrap(); + let mut corrupt = body.to_vec(); + corrupt[proofs[0].binding.locators[0].offset as usize + 8] ^= 1; + f.layout + .store() + .put_overwrite(&path, Bytes::from(corrupt)) + .await + .unwrap(); + let puts = f.count.put_requests(); + assert!(matches!( + f.directory + .select_node_bundle(&f.node, &prepared, &f.lease, Limits::default(), now) + .await, + Err(Error::Node("bundle frame digest differs")) + )); + assert_eq!( + f.count.put_requests(), + puts, + "corrupt historical frames select no new authority" + ); + f.layout.store().put_overwrite(&path, body).await.unwrap(); + // A prior selected proof and a proposal still cannot replace a fresh read + // of its historical dependency in a later operation. + f.count.block_body_reads_for(&path); + let puts = f.count.put_requests(); + assert!( + f.directory + .select_node_bundle(&f.node, &prepared, &f.lease, Limits::default(), now) + .await + .is_err() + ); + assert_eq!(f.count.put_requests(), puts); +} + +#[tokio::test] +async fn distinct_historical_reads_overlap_and_cancel_without_selecting_authority() { + use std::sync::atomic::Ordering; + let faults = Arc::new(super::super::faults::ReplyFault::default()); + let mut f = Fixture::with_store(faults.clone()).await; + super::super::coverage::enroll(&mut f).await; + let mut cells = [f.cell(4).await, f.cell(5).await]; + // Each original Cell frame lives in a different immutable object. + for cell in &mut cells { + let (_, frames, assigned) = f.append(cell, 2); + let proposal = f + .directory + .prepare_node_bundle(&f.node, &frames, &[assigned], NOW) + .await + .unwrap(); + f.node = f + .directory + .select_node_bundle(&f.node, &proposal, &f.lease, Limits::default(), NOW) + .await + .unwrap() + .0; + } + let mut frames = Vec::new(); + let mut assigned = Vec::new(); + for cell in &mut cells { + let (_, capture, range) = f.append(cell, 3); + frames.extend(capture); + assigned.push(range); + } + let proposal = f + .directory + .prepare_node_bundle(&f.node, &frames, &assigned, NOW) + .await + .unwrap(); + faults.mode.store(11, Ordering::SeqCst); + f.count.reset(); + { + let selection = + f.directory + .select_node_bundle(&f.node, &proposal, &f.lease, Limits::default(), NOW); + tokio::pin!(selection); + let overlap = async { + loop { + let started = faults.node_started.notified(); + if faults.range_started.load(Ordering::SeqCst) == 2 { + break; + } + started.await; + } + }; + tokio::select! { + _ = &mut selection => panic!("held historical reads must not select"), + result = tokio::time::timeout(std::time::Duration::from_secs(5), overlap) => result.unwrap(), + } + // Drop the pending original selection while both fresh reads wait. + } + assert_eq!( + f.count.put_requests(), + 0, + "cancellation selects no new authority" + ); + faults.mode.store(0, Ordering::SeqCst); + faults.node_resume.notify_waiters(); + let (_, proofs) = f + .directory + .select_node_bundle(&f.node, &proposal, &f.lease, Limits::default(), NOW) + .await + .unwrap(); + assert_eq!( + proofs.len(), + 2, + "a subsequent operation owns fresh scratch admission" + ); + assert!(proofs.iter().all(|proof| proof.commit_sequence() == 3)); +} diff --git a/crates/cellule-runtime/src/node/bundle/tests/index/mod.rs b/crates/cellule-runtime/src/node/bundle/tests/index/mod.rs index 79f7654d..5826a9c0 100644 --- a/crates/cellule-runtime/src/node/bundle/tests/index/mod.rs +++ b/crates/cellule-runtime/src/node/bundle/tests/index/mod.rs @@ -44,4 +44,5 @@ mod cohort; mod compatibility; mod copy_on_write; mod density; +mod history_cohort; mod inventory; diff --git a/crates/cellule-runtime/src/node/bundle/verification/mod.rs b/crates/cellule-runtime/src/node/bundle/verification/mod.rs new file mode 100644 index 00000000..98e11c96 --- /dev/null +++ b/crates/cellule-runtime/src/node/bundle/verification/mod.rs @@ -0,0 +1,262 @@ +//! Fresh cohort reads; exact frame and Cell-chain validation stays canonical. +use super::*; +use futures_util::{StreamExt, stream}; +use std::ops::Range; +use std::sync::Mutex; +use tokio::sync::Semaphore; + +const READ_CONCURRENCY: usize = 8; +// Together scratch and metadata replace the released fresh-origin buffer; +// they never add to the producer's original 20-MiB working reservation. +const SCRATCH_BYTES: u64 = MAX_BUNDLE_BYTES / 2; +#[cfg(test)] +const METADATA_BYTES: usize = (MAX_BUNDLE_BYTES / 2) as usize; + +#[derive(Clone, Copy)] +struct ReadIndex { + binding: u16, + locator: u16, +} + +struct Window { + object: Digest, + range: Range, + useful_bytes: u64, + reads: Vec, +} + +fn locator(bindings: &[Binding], read: ReadIndex) -> &Locator { + // Indices are constructed from these immutable slices inside this module. + &bindings[usize::from(read.binding)].locators[usize::from(read.locator)] +} + +struct Windows<'a> { + bindings: &'a [Binding], + reads: std::iter::Peekable>, +} + +fn windows(bindings: &[Binding]) -> Result> { + if bindings.len() > MAX_FRAMES + || bindings + .iter() + .any(|binding| binding.locators.len() > MAX_LOCATORS) + { + return Err(Error::Capacity("bundle cohort verification metadata")); + } + let count = bindings.iter().map(|binding| binding.locators.len()).sum(); + let mut reads = Vec::with_capacity(count); + for (binding, value) in bindings.iter().enumerate() { + for (index, value) in value.locators.iter().enumerate() { + value + .object + .ok_or(Error::Node("bundle locator is unresolved"))?; + value + .offset + .checked_add(value.bytes) + .ok_or(Error::Node("bundle locator overflow"))?; + if value.bytes == 0 || value.bytes > MAX_BUNDLE_BYTES { + return Err(Error::Capacity("bundle historical window bytes")); + } + reads.push(ReadIndex { + binding: u16::try_from(binding) + .map_err(|_| Error::Capacity("bundle cohort verification metadata"))?, + locator: u16::try_from(index) + .map_err(|_| Error::Capacity("bundle cohort verification metadata"))?, + }); + } + } + reads.sort_unstable_by_key(|read| { + let value = locator(bindings, *read); + (value.object.map(|object| *object.as_bytes()), value.offset) + }); + // Keep only compact sorted indices. Windows are assembled lazily, bounded + // by the reader concurrency rather than all 64 * 256 possible extents. + Ok(Windows { + bindings, + reads: reads.into_iter().peekable(), + }) +} + +impl Iterator for Windows<'_> { + type Item = Result; + fn next(&mut self) -> Option { + self.reads.next().map(|read| self.window(read)) + } +} + +impl Windows<'_> { + fn window(&mut self, read: ReadIndex) -> Result { + let value = locator(self.bindings, read); + let mut window = Window { + object: value + .object + .ok_or(Error::Node("bundle locator is unresolved"))?, + range: value.offset..value.offset + value.bytes, + useful_bytes: value.bytes, + reads: vec![read], + }; + while let Some(read) = self.reads.peek().copied() { + let value = locator(self.bindings, read); + if value.object != Some(window.object) { + break; + } + // Every end was checked before this immutable plan was formed. + let end = value.offset + value.bytes; + let useful = window + .useful_bytes + .checked_add(end.saturating_sub(window.range.end.max(value.offset))) + .ok_or(Error::Node("bundle locator overflow"))?; + let span = end.max(window.range.end) - window.range.start; + // Overlapping extents buy no padding; sparse windows cannot read + // more than twice the useful requested union or exceed scratch. + if span > SCRATCH_BYTES || span > useful.saturating_mul(2) { + break; + } + self.reads.next(); + window.range.end = end.max(window.range.end); + window.useful_bytes = useful; + window.reads.push(read); + } + Ok(window) + } +} + +pub(super) async fn verify_cohort( + layout: &cellule_ltx::CellStorageLayout, + session: SessionId, + epoch: u64, + bindings: &[Binding], + limits: cellule_ltx::Limits, + origin: &origin::OriginBundle, +) -> Result<()> { + if bindings.iter().any(|binding| { + binding + .locators + .iter() + .any(|locator| locator.bytes > SCRATCH_BYTES) + }) { + // A protocol-valid individual extent can exceed shared scratch. Keep + // the canonical serial path and allocate no cohort facts or windows. + for binding in bindings { + proof::verify_selected_binding(layout, session, epoch, binding, limits, origin).await?; + } + return Ok(()); + } + // Base traversal retains its original serial memory bound and fresh checks. + for binding in bindings { + proof::verify_base(layout, binding, limits).await?; + } + let windows = windows(bindings)?; + let facts = Mutex::new( + bindings + .iter() + .map(|binding| vec![None; binding.locators.len()]) + .collect::>(), + ); + let scratch = Semaphore::new(SCRATCH_BYTES as usize); + let mut reads = stream::iter(windows) + .map(|window| async { + read_window( + layout, session, epoch, bindings, limits, origin, &scratch, &facts, window?, + ) + .await + }) + .buffer_unordered(READ_CONCURRENCY); + while let Some(result) = reads.next().await { + result?; + } + drop(reads); + let facts = facts + .into_inner() + .map_err(|_| Error::Node("bundle cohort facts lock poisoned"))?; + for (binding, facts) in bindings.iter().zip(facts) { + let mut chain = proof::BindingChain::new(binding)?; + for step in facts { + chain.accept(step.ok_or(Error::Node("bundle cohort omits locator"))?)?; + } + chain.finish(binding)?; + } + Ok(()) +} + +#[allow( + clippy::too_many_arguments, + reason = "one read retains its exact scope and shared scratch admission" +)] +async fn read_window( + layout: &cellule_ltx::CellStorageLayout, + session: SessionId, + epoch: u64, + bindings: &[Binding], + limits: cellule_ltx::Limits, + origin: &origin::OriginBundle, + scratch: &Semaphore, + facts: &Mutex>>>, + window: Window, +) -> Result<()> { + let span = window.range.end - window.range.start; + let current = origin.range(session, epoch, window.object, &window.range)?; + let _permit = if current.is_none() { + Some( + scratch + .acquire_many( + u32::try_from(span) + .map_err(|_| Error::Capacity("bundle historical window bytes"))?, + ) + .await + .map_err(|_| Error::RuntimeClosed)?, + ) + } else { + None + }; + let bytes = match current { + Some(bytes) => bytes, + None => { + origin::read_range( + layout, + session, + epoch, + window.object, + window.range.clone(), + None, + ) + .await? + } + }; + if bytes.len() as u64 != span { + return Err(Error::Node("bundle extent is truncated")); + } + // Verification is synchronous in this one reader task. Only I/O overlaps; + // one decoder's scratch is live at a time, as on the original serial path. + // No body or duplicate vector of facts escapes this window. + let mut facts = facts + .lock() + .map_err(|_| Error::Node("bundle cohort facts lock poisoned"))?; + for read in window.reads { + let value = locator(bindings, read); + let start = usize::try_from(value.offset - window.range.start) + .map_err(|_| Error::Node("bundle extent overflow"))?; + let length = + usize::try_from(value.bytes).map_err(|_| Error::Node("bundle extent overflow"))?; + let frame = proof::checked_frame( + session, + epoch, + &bindings[usize::from(read.binding)], + value, + bytes.slice(start..start + length), + limits, + )?; + if facts[usize::from(read.binding)][usize::from(read.locator)] + .replace(proof::FrameStep::from_frame(&frame)) + .is_some() + { + return Err(Error::Node("bundle cohort repeats locator")); + } + } + // No native body escapes this operation; the shared scratch permit covers + // each fresh read until every frame in that window has been inspected. + Ok(()) +} + +#[cfg(test)] +mod tests; diff --git a/crates/cellule-runtime/src/node/bundle/verification/tests.rs b/crates/cellule-runtime/src/node/bundle/verification/tests.rs new file mode 100644 index 00000000..b2228f41 --- /dev/null +++ b/crates/cellule-runtime/src/node/bundle/verification/tests.rs @@ -0,0 +1,105 @@ +use super::*; +use crate::control::Owner; +use crate::identity::{ApplicationId, CellId, IncarnationId}; + +fn binding(locators: Vec) -> Binding { + Binding { + application: ApplicationId::from_bytes([9; 16]), + first_commit: 1, + control: Control::initial( + CellId::from_bytes([4; 32]), + IncarnationId::from_bytes([14; 16]), + Owner { + session: SessionId::from_bytes([1; 16]), + endpoint: "https://test.internal".into(), + }, + Digest::from_bytes([12; 32]), + 1, + ) + .unwrap(), + phase: BindingPhase::Open, + terminal: None, + selected_sequence: 1, + selected_commit: 1, + selected_position: cellule_ltx::Position { + txid: 1, + checksum: 2, + }, + locators, + } +} + +fn extent(object: u8, offset: u64, bytes: u64) -> Locator { + Locator { + object: Some(Digest::from_bytes([object; 32])), + offset, + bytes, + frame_digest: Digest::from_bytes([7; 32]), + } +} + +#[test] +fn sparse_ranges_bound_padding_use_union_bytes_and_preserve_every_locator() { + let bindings = [binding(vec![ + extent(1, 0, 10), + extent(1, 5, 10), + extent(1, 25, 10), + extent(1, 500, 10), + extent(2, 0, 10), + ])]; + let planned = windows(&bindings) + .unwrap() + .collect::>>() + .unwrap(); + assert_eq!(planned.len(), 3); + assert_eq!(planned[0].range, 0..35); + assert_eq!( + planned[0].useful_bytes, 25, + "overlap cannot buy extra padding" + ); + assert_eq!(planned[1].range, 500..510); + assert_eq!(planned[2].object, Digest::from_bytes([2; 32])); + assert_eq!( + planned + .iter() + .map(|window| window.reads.len()) + .sum::(), + 5 + ); + for window in &planned { + assert!(window.range.end - window.range.start <= 2 * window.useful_bytes); + } + let large = [binding(vec![ + extent(1, 0, SCRATCH_BYTES), + extent(1, SCRATCH_BYTES, 1), + ])]; + assert_eq!( + windows(&large).unwrap().count(), + 2, + "a window never exceeds shared scratch" + ); +} + +#[test] +fn protocol_maximum_verification_payload_fits_its_admitted_metadata_charge() { + // Bound eight lazy windows, original compact indices, growing window-read + // capacities including a reallocation, and one table of checked facts. + // Leave allocator/task overhead inside the remaining admitted headroom. + let count = MAX_FRAMES * MAX_LOCATORS; + let windows = READ_CONCURRENCY * std::mem::size_of::(); + let indices = 4 * count * std::mem::size_of::(); + let facts = count * std::mem::size_of::>(); + let rows = MAX_FRAMES * std::mem::size_of::>>(); + let total = windows + indices + facts + rows; + assert!( + total + 256 * 1024 <= METADATA_BYTES, + "payload {total} leaves less than 256 KiB overhead" + ); + let invalid = [binding(vec![extent(1, 0, 0)])]; + assert!(matches!(super::windows(&invalid), Err(Error::Capacity(_)))); + let invalid = [binding(vec![extent(1, u64::MAX, 1)])]; + assert!(matches!( + super::windows(&invalid), + Err(Error::Node("bundle locator overflow")) + )); +} diff --git a/crates/cellule-runtime/src/node/durability/publication/mod.rs b/crates/cellule-runtime/src/node/durability/publication/mod.rs index 1a2956ad..eed86100 100644 --- a/crates/cellule-runtime/src/node/durability/publication/mod.rs +++ b/crates/cellule-runtime/src/node/durability/publication/mod.rs @@ -81,8 +81,9 @@ impl Publisher { .ok_or(Error::PendingPublication)?; resources.try_reserve( crate::fleet::resource::ResourceCost::zero() - // The fifth bounded buffer is the fresh cohort origin read; - // its bytes are shared only within one verification operation. + // The fifth buffer covers the fresh cohort origin read. After + // matching the proposal, its allocation is released for 2 MiB + // of historical scratch and bounded operation-local facts. .with_retained_bytes((5 * crate::node::bundle::MAX_BUNDLE_BYTES) as usize), ) } From 3f459367d059207891218dcb14a031ab1a086e8d Mon Sep 17 00:00:00 2001 From: forhappy Date: Fri, 9 Oct 2026 00:19:21 -0700 Subject: [PATCH 062/102] Record bounded history measurements and revised node target --- .../docs/write-performance-design.md | 58 +++--- docs/bundle-coverage-implementation.md | 24 ++- docs/pr67-bounded-history-measurement.md | 183 ++++++++++++++++++ 3 files changed, 227 insertions(+), 38 deletions(-) create mode 100644 docs/pr67-bounded-history-measurement.md diff --git a/crates/cellule-runtime/docs/write-performance-design.md b/crates/cellule-runtime/docs/write-performance-design.md index 7e0ba284..231d28ab 100644 --- a/crates/cellule-runtime/docs/write-performance-design.md +++ b/crates/cellule-runtime/docs/write-performance-design.md @@ -1,22 +1,19 @@ # Node write and read performance design -The [submission-timing diagnostic](../../../docs/pr67-submission-timing-measurement.md) -now measures the missing pre-proof queue: 727.30 ms waiting for the global -issuance lock, whose holder waits 6.14 ms for publication capacity. The -instrumented application completes 158.37 writes/s, with all 19,675 ACKs passing -warm/cold audit. This is diagnosis, not a throughput improvement or qualification. +The [bounded historical-read comparison](../../../docs/pr67-bounded-history-measurement.md) +now groups fresh ranges within the original 20-MiB admission. One original-profile +Fleet pair completes 307.60 writes/s versus 144.50 before, with worse successful +p99 latency. A matched 1-GiB control reaches 237.67 versus 181.45/s, with zero +request errors. All Cellule ACKs pass warm/cold audit. These short overloaded +observations remain unqualified; PR #67 is still a draft. -The [latest replay-pressure verification](../../../docs/pr67-replay-admission-measurement.md) -preserves original durable outcomes under publication pressure. One fresh Fleet -pair completes 158.15 writes/s versus 152.92 before, with all 19,542 candidate ACKs -passing warm/cold audit and joined drain. A separate 1-GiB budget control reaches -only 191.08 writes/s; its latency worsens versus the matching baseline. Neither -establishes repeatable performance improvement or parity. Native ticket issuance -still waits for the bounded publication lane before follower proof can begin, -and historical/base verification remains expensive. Resource policies in the -celld comparison are asymmetric. The -[historical-read experiment](../../../docs/pr67-historical-read-measurement.md) -remains withdrawn after its severe application regression. +The [submission diagnosis](../../../docs/pr67-submission-timing-measurement.md) +identified publication capacity held under the global issuance lock. That +coupling still queues native progress before follower proof starts. Repeated +historical/base verification and sparse root checkpoints keep the publication +consumer expensive. Matching celld requires reducing that work and separating +native progress from bounded recoverable publication debt; larger queues alone +do not increase sustainable throughput. Status: implementation in progress. The application path is not qualified at the targets below. Component I/O reductions are not application TPS. @@ -32,15 +29,15 @@ The embedding application continues to own ingress and authorization. | --- | --- | | Serving node | 8 vCPUs, 16 GiB memory; report storage, filesystem, network and SQLite policy | | Population | 2,000 uniformly active Cells, sub-100-byte values | -| Writes | 10,000 successful commands/s | -| Reads | 50,000 successful queries/s | +| Writes | 2,000 successful commands/s | +| Reads | 20,000 successful queries/s | | Tail latency | Fleet writes p99 at most 50 ms; Bucket writes p99 at most 200 ms; report read p50/p95/p99 | | Delivery | Zero errors, dropped offers or unissued requests; at least 99% completed within the window | | Evidence | Three paired repetitions of at least five minutes, matched celld revision and workload | | Durability | All acknowledged mutations and retry outcomes survive warm audit, joined drain and cold restore | | Stability | Bounded memory, native-job occupancy and debt; debt has no sustained positive slope | -Qualify read-only, write-only and simultaneous 10K-write/50K-read load separately. +Qualify read-only, write-only and simultaneous 2K-write/20K-read load separately. Count the client, provider and followers separately from the serving node. Docker can simulate the deployment, but sharing one 8-CPU VM between all roles does not qualify an 8-CPU serving node. Report KV overwrite and SQL commands with @@ -83,9 +80,13 @@ Each selection reads its complete new cohort object once from origin, compares every byte with the proposal, then verifies header, shards, histories and native frames from that operation's read. It retains no cross-operation availability cache. Historical objects and every Cell base dependency still require origin -verification. Selection drops each checked native frame rather than retaining -reconstruction bodies. The additional 4-MiB buffer is charged before installing -the producer; workload retention and protocol bounds remain unchanged. +verification. After the fresh body matches, selection shares the proposal +allocation and uses its released buffer allowance for 2 MiB of historical scratch +and at most 2 MiB of compact facts/planning metadata. Eight bounded reads overlap; +every frame and original Cell chain remains canonically checked. Base traversal +and individual extents above 2 MiB retain serial verification. Selection retains +no reconstruction bodies beyond the operation. Working admission remains 20 MiB; +workload retention and protocol bounds remain unchanged. The first end-to-end Fleet diagnostic of this connection failed throughput, availability and drain. It is experimental, not performance qualification. @@ -187,9 +188,16 @@ independently addressed locator histories: Those are protocol bounds, not host admission. The materializer scheduler must charge retained history, native verification, scratch and outcomes to the node -ledgers. Uniform 10K writes over 2,000 Cells means about five commands/Cell/s: -215 commands span about 43 seconds. Measure the actual native bytes retained -over that interval and reject the model if the 16-GiB node cannot hold its debt. +ledgers. At the revised 2K-write target over 2,000 uniform Cells, each Cell +receives about one command/s: 215 commands span about 215 seconds. The current +45-second root-age trigger therefore requests a checkpoint at roughly 45 +commands even before byte pressure. The conditional four-PUT model then costs +about 0.121 PUTs/command, above 0.05. Meeting that separate cost target requires +a measured change to checkpoint scheduling or publication representation; +increasing locator capacity alone cannot meet it. Measure actual retained native +bytes and oldest debt before changing the age policy, and reject a policy that +cannot stay bounded on the 16-GiB node. Earlier benchmark profiles retain their +original thresholds and evidence. ## Remaining implementation and exit gates diff --git a/docs/bundle-coverage-implementation.md b/docs/bundle-coverage-implementation.md index 7c747e5c..bb8cae6e 100644 --- a/docs/bundle-coverage-implementation.md +++ b/docs/bundle-coverage-implementation.md @@ -1,18 +1,16 @@ # Bundle coverage implementation -The [submission-timing diagnostic](pr67-submission-timing-measurement.md) -attributes the main pre-proof delay to publication capacity held under the global -issuance lock: 727.30 ms mean lock wait, with a 6.14-ms publication wait inside -the lock. Its 158.37 writes/s result is unqualified; all 19,675 ACKs pass warm/cold -audit. No throughput gain is claimed for this instrumentation slice. - -The [latest replay-pressure diagnostic](pr67-replay-admission-measurement.md) -preserves original durable command outcomes under node and Cell publication -pressure. One fresh Fleet pair completes 158.15 writes/s versus 152.92 before -and 4,473.32 for celld. Candidate warm/cold audits pass for all 19,542 ACKs; -all three fail performance qualification. A separate 1-GiB retained-credit control -reaches only 191.08 writes/s. Resource policies are asymmetric, and no repeatable -throughput improvement or parity is established. PR #67 remains a draft. +The [bounded historical-read comparison](pr67-bounded-history-measurement.md) +groups fresh ranges within the original 20-MiB admission. One original-profile +pair completes 307.60 writes/s versus 144.50 before, with worse successful p99. +A 1-GiB control with zero request errors reaches 237.67 versus 181.45/s. +Complete Cellule warm/cold ACK audits pass, but all capacity profiles remain +unqualified. + +The [submission diagnosis](pr67-submission-timing-measurement.md) identifies +publication capacity held under the global issuance lock. The current selector +still gates native progress before follower proof. Celld pipelines that progress +independently of bucket publication. PR #67 remains a draft. The installed original producer provides admitted shared receipts, independent root materialization and complete live-writer closure. The Fleet SQL example diff --git a/docs/pr67-bounded-history-measurement.md b/docs/pr67-bounded-history-measurement.md new file mode 100644 index 00000000..e364aa35 --- /dev/null +++ b/docs/pr67-bounded-history-measurement.md @@ -0,0 +1,183 @@ +# PR 67: historical reads within original admission + +Candidate `90f099a295158c6effe51079d35db07eab204d17` groups fresh historical +ranges without increasing the original 20-MiB publication reservation. In one +original-budget Fleet pair, completed writes rise from **144.50 to 307.60/s**, +but successful tail latency worsens. A separately matched 1-GiB control reaches +237.67/s versus 181.45 before, with zero request errors and passing complete +warm/cold audits. These short overloaded diagnostics do not establish parity or +repeatable capacity. PR #67 remains a draft. + +## Delivered change and reproduction + +Selection still reads the complete new cohort from origin and compares every +byte with its proposal. After that match, it shares the proposal allocation and +reuses the released origin-buffer allowance for 2 MiB of historical read scratch +and at most 2 MiB of checked facts and planning metadata. + +At most eight historical reads overlap. Compact indices group contiguous or +nearby ranges from the same object; padding cannot exceed useful union bytes. +Each frame keeps canonical digest, native envelope and Cell-scope verification. +Checked facts are validated in the original per-Cell order, including exact +command, transaction, checksum and final endpoints. No body or availability +cache survives the selection operation. Base dependencies remain freshly +verified on the original serial path. Individual extents above 2 MiB also use +the canonical serial verifier, avoiding an oversized semaphore acquisition. + +The real 64-Cell regression fails in three baseline executions with 64 historical +reads; the candidate needs one. A separate held-read regression reproduces +serial I/O before and verifies overlap and cancellation after. Corrupt or missing +historical dependencies still prevent new authority selection. A deterministic +large-extent test verifies the serial route and exact cold payload/outcome +recovery. Maximum metadata, including vector growth and 256-KiB headroom, fits +2 MiB at 64 Cells × 256 locators. The existing 20-MiB receipt-pressure and FIFO +contracts pass unchanged. This does not restore the earlier experiment's extra +3-MiB admission, which caused its severe application regression. + +## Fresh original-budget comparison + +Baseline is `8bde901c7b5237c5f28f1da70231f1283f0f1b0f`; celld is +`f2bf648663a610eefde71f3547ad61e9b896b1f0`. All cases use the same client, +auditor, fixtures and images: 1,000 uniform Cells, 96-byte SQL values with a +two-hour outcome ledger, 128 clients/queue slots, 15K offered writes/s, +30-second warmup and a 60-second window. Before/after request configuration and +execution metadata match, apart from source/binary and fresh namespace identity. +No build or contributor suite overlaps a timed window. + +| System | Completed writes/s | Successful scheduled p99 ms | Successful request p99 ms | Errors | Queue drops | +| --- | ---: | ---: | ---: | ---: | ---: | +| Before | 144.50 | 837.55 | 814.04 | 349,995 | 541,292 | +| Candidate | 307.60 | 1,690.07 | 950.06 | 201,942 | 679,542 | +| celld | 4,622.43 | 144.82 | 73.54 | 0 | 622,398 | + +TPS counts successes inside the window. Successful percentiles include trailing +measured successes and exclude errors/drops; scheduled time begins at offered +arrival. The independent replay also reconciles all attempt and drop counts. +Candidate has 18,456 in-window and 60 trailing successes; baseline has 8,670 and +43. Warmup request errors are 128 after versus 847 before. The 112.87% completion +difference is a single-pair observation; both successful latency measures worsen. +All-attempt percentiles are lower because they include many quick refusals and +cannot substitute for successful-command latency or zero-drop qualification. + +| System | Complete ACK cohort | Warm audit | Cold audit | Joined drain s | +| --- | ---: | --- | --- | ---: | +| Before | 18,438 | all reads/retries pass | all reads/retries pass | 27.60 | +| Candidate | 33,817 | all reads/retries pass | all reads/retries pass | 27.21 | +| celld | 455,122 | all reads/retries pass | not reached | unverified | + +Celld's owner is OOM-killed during shutdown, exit 137, after its passing warm +audit. Missing cold/drain gates do not themselves prove mutation loss. Cellule's +provider health and lifecycle observations pass; celld's cold/final provider +observations are incomplete. Every complete ACK count and journal reconciles. + +## Matched larger-budget control + +This pair changes only retained-work credit from 64 MiB to 1 GiB. Managed disk +remains 1 GiB; it uses the same workload, binaries and VM. Candidate ran before +this pair's fresh baseline. It is a separate diagnostic, not a substituted +qualification profile or matched celld resource-policy comparison. + +| Code | Completed writes/s | Successful scheduled p99 ms | Successful request p99 ms | Errors | Queue drops | Complete warm/cold ACK cohort | Drain s | +| --- | ---: | ---: | ---: | ---: | ---: | ---: | ---: | +| Before | 181.45 | 2,523.49 | 1,473.94 | 0 | 888,857 | 19,924; all pass | 30.37 | +| Candidate | 237.67 | 2,348.86 | 1,459.35 | 0 | 885,484 | 29,284; all pass | 31.42 | + +Completion rises 30.98%; scheduled p99 improves about 6.9%, while request p99 +changes little. Both pairs remain overloaded, short and unqualified. These +observations support reducing selection work; they do not establish a repeatable +throughput/latency gain or meet any requested capacity target. + +## Cost and remaining bottleneck + +Original-profile GET/range attempts per in-window completion fall from 13.95 to +9.38; successful PUTs fall from 0.626 to 0.355. Materialized commands per root +rise from 11.72 to 16.90. The API counters include background cohorts, exclude +SDK retries and have window boundaries, so they are not exact per-command costs. +Total API work can increase as more writes complete. The conditional +215-command / 0.05-PUT target remains unmet. + +In the error-free control, exact submission partitions are: + +| Phase | Before mean ms | Candidate mean ms | +| --- | ---: | ---: | +| Local load | 0.193 | 0.270 | +| Ordered issuance lock | 651.079 | 485.908 | +| Publication capacity, with lock held | 5.386 | 4.141 | +| Assignment/enqueue | 0.00695 | 0.00851 | +| Complete submission | 656.665 | 490.328 | + +Native credit, shipping-slot and validation means are below 0.001 ms. Every +phase count matches its successful assignment cohort and durations sum exactly. +The cohorts differ from in-window HTTP response counts; other worker/proof +timers are not additive to them. In the original pressure pair, mean lock wait +instead rises from 163.63 to 213.15 ms as the admitted population changes. Do not +combine these two budget profiles into a single latency conclusion. + +The selector still gates the global native lane. Fresh base traversal remains +serial; repeated historical body verification, index work and root publication +remain expensive. Native shipping awaits one complete append round before the +next. Next work must bound and reduce those costs, preserve authenticated prefix +and availability witnesses, and separate native progress from recoverable +publication debt. Queue enlargement alone cannot increase sustainable selection +throughput. Failed-owner suffix recovery, safe collection, Bucket integration +and read/mixed qualification still require their complete evidence. + +## Why the architecture still differs from celld + +Both systems use per-Cell SQLite, LTX capture, fenced ownership and follower logs. +Their dependencies before a Fleet acknowledgement differ: + +| Concern | Cellule candidate | Pinned celld reference | +| --- | --- | --- | +| Admission before native issuance | Global ordered lock reserves bounded publication capacity before assigning the ticket | Fleet capture/shipping loop proceeds independently of bucket waits | +| Follower rounds | A complete append round finishes before the next starts | Ordered member lanes allow multiple rounds in flight; credits apply in submission order | +| Follower commit | Canonical batched append and durable watermark | Ordered stream groups already delivered frames into one durable append batch | +| Publication work | Fresh bases and historical ranges are verified again; about 9.38 GET/range attempts per completed write in this window | Separate node-bundle upload and coverage work; the Fleet loop does not wait for it | +| Local SQLite policy | Managed sessions use WAL `NORMAL` with external durability proofs | Steady-state Cell connections also use WAL `NORMAL` | + +The [Cellule issuance path](../crates/cellule-runtime/src/node/log_shipper/mod.rs) +holds `order` across `publication.reserve`; its shipper awaits each `append_batch`. +Celld's [Fleet loop](https://github.com/denoland/celld/blob/f2bf648663a610eefde71f3547ad61e9b896b1f0/crates/celld/ltx_repl.rs#L4514) +uses ordered in-flight rounds and explicitly avoids bucket waits. Its +[follower stream](https://github.com/denoland/celld/blob/f2bf648663a610eefde71f3547ad61e9b896b1f0/crates/celld/node_log.rs#L160) +groups delivered frames before the durable append. Its +[Cell storage setup](https://github.com/denoland/celld/blob/f2bf648663a610eefde71f3547ad61e9b896b1f0/crates/celld/storage.rs#L444) +uses WAL `NORMAL`; Cellule's managed sessions also select that policy. These +source differences explain the mechanism; they do not assign a measured +fraction of the total TPS gap to every difference. + +In the earlier exact submission diagnostic, a 6.14-ms publication wait plus +0.007-ms assignment while holding the global lock implies about 163 serial +submissions/s, consistent with its 158.37 completed writes/s. This is a queueing +interpretation of that run, not a capacity forecast. The latest range grouping +reduces publication I/O, but leaves the dependency itself in place. Fleet ACKs +still need a complete recoverable issued range, bounded debt and joined drain; +simply bypassing publication admission would violate those contracts. + +## Verification and limitations + +All contributor routes and Rust 1.99 Clippy pass in an immutable source snapshot: +1,970 workspace tests and 60 local LTX tests pass; 38 environment-dependent tests +remain ignored. The new regressions pass three times after correction. All 1,299 +Rust/Cargo files match that snapshot and the pinned production build source. + +The shared ARM64 Docker VM has 8 CPUs and 8 GiB total RAM. Container ceilings +are 8 CPUs/16 GiB per node, with 4-GiB tmpfs, and exceed aggregate VM resources. +Only Cellule receives explicit retained-work and managed-disk limits; celld +metadata does not apply equivalent internal limits. No case qualifies a dedicated +8-vCPU/16-GiB node or physical-media durability. All five performance reports +fail qualification. There is no new Bucket or read-only measurement. The +revised 2,000-Cell / 2K-write / 20K-read objective and three matched five-minute +repetitions remain unqualified. + +Raw sources, failed baseline tests, binaries, fixtures, verification, journals, +metrics, reconciliation and identity checks stay outside Git under +`/Volumes/Workspace/crabbuild-target/cellule-write-perf-8ad1/bounded-history-20261008-*`, +including the two new baseline cases under the reused instrumented build. +The external evidence index and rehash certificate cover the retained inventory +and reused dependencies. No raw performance journal is committed. + +[Submission bottleneck](pr67-submission-timing-measurement.md), +[withdrawn earlier experiment](pr67-historical-read-measurement.md), +[implementation](bundle-coverage-implementation.md), +[capacity contract](../crates/cellule-runtime/docs/write-performance-design.md). From 29b94157c5915819c65b5ee80a345c2f32a41f3c Mon Sep 17 00:00:00 2001 From: forhappy Date: Fri, 9 Oct 2026 01:00:11 -0700 Subject: [PATCH 063/102] Overlap bounded fresh Cell base verification --- crates/cellule-ltx/api-prelude.txt | 1 + crates/cellule-ltx/docs/publication.md | 23 +++ crates/cellule-ltx/src/lib.rs | 6 +- crates/cellule-ltx/src/replica/mod.rs | 2 + crates/cellule-ltx/src/replica/origin/mod.rs | 104 +++++++++++++ .../cellule-ltx/src/replica/origin/tests.rs | 20 +++ crates/cellule-ltx/src/replica/verify.rs | 33 +++- crates/cellule-ltx/tests/cell/roots.rs | 1 + crates/cellule-ltx/tests/cell/roots/origin.rs | 106 +++++++++++++ .../cellule-runtime/src/node/bundle/proof.rs | 32 +++- .../src/node/bundle/tests/faults.rs | 9 ++ .../node/bundle/tests/index/base_cohort.rs | 142 ++++++++++++++++++ .../src/node/bundle/tests/index/mod.rs | 1 + .../src/node/bundle/verification/base.rs | 50 ++++++ .../src/node/bundle/verification/mod.rs | 8 +- scripts/perf/README.md | 7 +- scripts/perf/run.py | 14 +- 17 files changed, 537 insertions(+), 22 deletions(-) create mode 100644 crates/cellule-ltx/src/replica/origin/mod.rs create mode 100644 crates/cellule-ltx/src/replica/origin/tests.rs create mode 100644 crates/cellule-ltx/tests/cell/roots/origin.rs create mode 100644 crates/cellule-runtime/src/node/bundle/tests/index/base_cohort.rs create mode 100644 crates/cellule-runtime/src/node/bundle/verification/base.rs diff --git a/crates/cellule-ltx/api-prelude.txt b/crates/cellule-ltx/api-prelude.txt index afe4a796..8d76a028 100644 --- a/crates/cellule-ltx/api-prelude.txt +++ b/crates/cellule-ltx/api-prelude.txt @@ -38,6 +38,7 @@ ReadOnlyRoot RecoveryOverlay Result RootObjectRef +RootOriginVerification RootPreparation RootPreparationFuture RootPreparationMetadata diff --git a/crates/cellule-ltx/docs/publication.md b/crates/cellule-ltx/docs/publication.md index e69f9166..32668a53 100644 --- a/crates/cellule-ltx/docs/publication.md +++ b/crates/cellule-ltx/docs/publication.md @@ -146,3 +146,26 @@ Directory digests obey that bound while descriptor work keeps the existing fixed root/segment ceilings. The caller owns memory admission, bounded Store stream chunks and the enclosing deadline. This graph proof grants no selected authority, retention pin or current serving. + +`small_root_origin_verification` begins the same origin walk with a fresh exact +root read. Packed leaf graphs with at most 32 inline descriptors return a +one-use `RootOriginVerification`. Its 512-KiB working charge includes metadata, +one bounded body and verification scratch. The host owns admission and scheduling; +the root/decode phase also requires admission before starting. `verify` consumes +the operation, freshly checks every dependency through the canonical verifier +and returns the complete inventory. No body or availability cache survives it. +Other graph shapes return `None` and require the original complete traversal. + +```rust +use cellule_ltx::{CellReplica, RootObjectRef, RootRef}; + +async fn origin_inventory( + replica: &CellReplica, + root: &RootRef, +) -> cellule_ltx::Result> { + match replica.small_root_origin_verification(root, 65_536).await? { + Some(operation) => operation.verify().await, + None => replica.reachable_objects_bounded(root, 65_536).await, + } +} +``` diff --git a/crates/cellule-ltx/src/lib.rs b/crates/cellule-ltx/src/lib.rs index a3791778..3f810f4f 100644 --- a/crates/cellule-ltx/src/lib.rs +++ b/crates/cellule-ltx/src/lib.rs @@ -81,9 +81,9 @@ pub use node_frame::{ #[cfg(feature = "replica")] pub use replica::{ CellPagedDatabase, CellReplica, CellWritableDatabase, PreparedRoot, PublicationCost, - ReadOnlyRoot, RecoveryOverlay, RootObjectRef, RootPreparation, RootPreparationFuture, - RootPreparationMetadata, RootRef, SHARED_PUBLICATION_BYTES, SHARED_PUBLICATION_ROWS, - SharedAppend, SharedCaptures, VerifiedRoot, + ReadOnlyRoot, RecoveryOverlay, RootObjectRef, RootOriginVerification, RootPreparation, + RootPreparationFuture, RootPreparationMetadata, RootRef, SHARED_PUBLICATION_BYTES, + SHARED_PUBLICATION_ROWS, SharedAppend, SharedCaptures, VerifiedRoot, }; #[cfg(feature = "replica")] pub use writable_vfs::Hydration; diff --git a/crates/cellule-ltx/src/replica/mod.rs b/crates/cellule-ltx/src/replica/mod.rs index 11374ceb..384fa32e 100644 --- a/crates/cellule-ltx/src/replica/mod.rs +++ b/crates/cellule-ltx/src/replica/mod.rs @@ -24,11 +24,13 @@ pub use shared::{SHARED_PUBLICATION_BYTES, SHARED_PUBLICATION_ROWS, SharedAppend mod preparation; mod prepare; pub use preparation::{RootPreparation, RootPreparationFuture, RootPreparationMetadata}; +mod origin; mod read_only; mod restore; pub(crate) mod root; mod upload; mod verify; +pub use origin::RootOriginVerification; use directory::{DirectoryEntry, DirectorySpan, DirectoryTree, ObjectExtent}; pub use read_only::ReadOnlyRoot; diff --git a/crates/cellule-ltx/src/replica/origin/mod.rs b/crates/cellule-ltx/src/replica/origin/mod.rs new file mode 100644 index 00000000..4de63461 --- /dev/null +++ b/crates/cellule-ltx/src/replica/origin/mod.rs @@ -0,0 +1,104 @@ +//! One fresh small-root verification operation; scheduling stays with the host. +use super::*; +use std::time::Instant; + +const WORKING_BYTES: usize = 512 << 10; + +/// A single origin verification operation begun with a fresh exact root read. +/// +/// This retains bounded root metadata, not verified dependency availability. +/// Complete it before using the inventory; no writer, retention or durability +/// authority is granted. It is consumed once and cannot substitute a cached +/// proof for fresh reads in a later operation. Scheduling and memory admission +/// belong to the embedding runtime. +#[must_use = "complete origin verification before relying on the dependency inventory"] +pub struct RootOriginVerification { + replica: CellReplica, + root: RootRef, + document: RootDocument, + max_objects: usize, + started: Instant, +} + +impl CellReplica { + /// Begins fresh verification of a root with a bounded small working set. + /// + /// Reads and authenticates the exact root record without the metadata cache. + /// A packed leaf graph with at most 32 inline descriptors returns a one-use + /// operation. Other graphs return `None`; callers must use the complete + /// [`Self::reachable_objects_bounded`] path with its original admission. + /// No partial inventory or availability proof is returned. + pub async fn small_root_origin_verification( + &self, + root: &RootRef, + max_objects: usize, + ) -> Result> { + if max_objects == 0 { + return Err(LtxError::Limit(crate::LimitKind::RootInventoryObjects)); + } + let started = self.host.now_monotonic(); + let (document, _) = self.read_root_document(root, false).await?; + if !eligible(&document) { + return Ok(None); + } + Ok(Some(RootOriginVerification { + replica: self.clone(), + root: *root, + document, + max_objects, + started, + })) + } +} + +fn eligible(document: &RootDocument) -> bool { + document.segment_pages.is_empty() + && document.directory_height == 0 + && document.segments.len() <= MAX_INLINE_SEGMENTS + && document.segments.iter().all(|descriptor| { + matches!( + descriptor.object_kind(), + CellObjectKind::Packed | CellObjectKind::SharedPacked + ) + }) +} + +impl RootOriginVerification { + /// Working-set charge including metadata, one small origin body and scratch. + /// + /// Retained root/descriptor/inventory tables and a leaf directory are bounded + /// independently of database size. Only one packed body, at most 256 KiB, + /// is read at a time. Hosts must admit this charge before overlapping work. + #[must_use] + pub const fn working_bytes(&self) -> usize { + WORKING_BYTES + } + + /// Authenticates every origin dependency and returns the complete inventory. + /// + /// Uses the canonical graph, packed-body, directory and inventory verifier. + /// Any missing, corrupt or out-of-scope dependency fails with its source + /// error. Bodies do not survive this operation; cancellation drops pending + /// origin reads and the caller must retain its admission through completion. + pub async fn verify(self) -> Result> { + let Self { + replica, + root, + document, + max_objects, + started, + } = self; + let graph = replica + .load_graph_document(&root, false, document, None) + .await; + replica + .host + .observe_ltx_phase(crate::LtxPhase::RootOpen, started, graph.is_ok()); + replica + .inventory_graph(&root, Some(max_objects), graph?) + .await + } +} + +#[cfg(test)] +mod tests; diff --git a/crates/cellule-ltx/src/replica/origin/tests.rs b/crates/cellule-ltx/src/replica/origin/tests.rs new file mode 100644 index 00000000..cc4071a4 --- /dev/null +++ b/crates/cellule-ltx/src/replica/origin/tests.rs @@ -0,0 +1,20 @@ +use super::*; + +#[test] +fn packed_leaf_working_payload_leaves_bounded_allocator_and_task_headroom() { + // Account for wire decoding/read copies, two descriptor vectors, extent + // and inventory node overhead, directory verification and shared rows. + let metadata = MAX_INLINE_SEGMENTS * (3 * std::mem::size_of::() + 256) + + 32 * 1024 + + shared::SHARED_PUBLICATION_ROWS * 128; + // Root wire decoding, directory scratch and the packed body occupy + // separate phases; directory frames do not survive into body reads. + let root_read = 4 * ROOT_BYTES as usize + metadata; + let directory = 3 * 32 * 1024 + metadata; + let body = upload::SINGLE_PUT_BYTES as usize + metadata; + let payload = root_read.max(directory).max(body); + assert!( + payload + 128 * 1024 <= WORKING_BYTES, + "working payload {payload} leaves less than 128 KiB overhead" + ); +} diff --git a/crates/cellule-ltx/src/replica/verify.rs b/crates/cellule-ltx/src/replica/verify.rs index 4ace5c28..dc4d0dd1 100644 --- a/crates/cellule-ltx/src/replica/verify.rs +++ b/crates/cellule-ltx/src/replica/verify.rs @@ -50,6 +50,15 @@ impl CellReplica { ) -> Result> { // Inventory must prove origin presence even for metadata uploaded here. let graph = self.load_graph_with_cache(root, false).await?; + self.inventory_graph(root, max_objects, graph).await + } + + pub(super) async fn inventory_graph( + &self, + root: &RootRef, + max_objects: Option, + graph: LoadedGraph, + ) -> Result> { let extents = object_extents(&graph.descriptors)?; let verification = directory::Verification { layout: &self.layout, @@ -230,6 +239,16 @@ impl CellReplica { } async fn load_graph_inner(&self, root: &RootRef, use_cache: bool) -> Result { + let (document, cached_root) = self.read_root_document(root, use_cache).await?; + self.load_graph_document(root, use_cache, document, cached_root) + .await + } + + pub(super) async fn read_root_document( + &self, + root: &RootRef, + use_cache: bool, + ) -> Result<(RootDocument, Option)> { self.check_scope(root)?; let (bytes, cached_root) = self .read_object(&root.digest, CellObjectKind::Root, ROOT_BYTES, use_cache) @@ -251,6 +270,16 @@ impl CellReplica { { return Err(LtxError::LTXCorrupted); } + Ok((document, cached_root.then_some(bytes.len()))) + } + + pub(super) async fn load_graph_document( + &self, + root: &RootRef, + use_cache: bool, + document: RootDocument, + cached_root: Option, + ) -> Result { let pages = stream::iter( document .segment_pages @@ -274,8 +303,8 @@ impl CellReplica { .try_collect::>() .await?; let mut cached_metadata = Vec::new(); - if cached_root { - cached_metadata.push((root.digest, bytes.len())); + if let Some(bytes) = cached_root { + cached_metadata.push((root.digest, bytes)); } let mut descriptors = Vec::new(); for (page, cached) in pages { diff --git a/crates/cellule-ltx/tests/cell/roots.rs b/crates/cellule-ltx/tests/cell/roots.rs index 81c9a48f..97d15505 100644 --- a/crates/cellule-ltx/tests/cell/roots.rs +++ b/crates/cellule-ltx/tests/cell/roots.rs @@ -37,6 +37,7 @@ mod compaction; mod compaction_transfers; mod directory; mod lifecycle; +mod origin; mod packed; mod preparation; mod prepare_cost; diff --git a/crates/cellule-ltx/tests/cell/roots/origin.rs b/crates/cellule-ltx/tests/cell/roots/origin.rs new file mode 100644 index 00000000..b880ee23 --- /dev/null +++ b/crates/cellule-ltx/tests/cell/roots/origin.rs @@ -0,0 +1,106 @@ +use super::*; + +#[tokio::test] +async fn one_use_origin_verification_keeps_the_complete_inventory_and_fresh_body_checks() { + let temp = tempfile::tempdir().unwrap(); + let mut db = Db::open(&temp.path().join("source"), Limits::default()).unwrap(); + db.transaction(|tx| tx.execute_batch("CREATE TABLE t(v); INSERT INTO t VALUES(7)")) + .unwrap(); + let cuts = db.capture().unwrap(); + let backend = Arc::new(InMemory::new()); + let store = Store::new(backend.clone()); + let layout = CellStorageLayout::new(store.clone(), Path::from("runtime"), [3; 16]); + let replica = replica(store, [161; 32], [162; 16]); + let root = replica.prepare(None, &cuts, 1, 1).await.unwrap().root(); + let expected = replica.reachable_objects_bounded(&root, 64).await.unwrap(); + let operation = replica + .small_root_origin_verification(&root, 64) + .await + .unwrap() + .unwrap(); + assert_eq!(operation.working_bytes(), 512 << 10); + assert_eq!(operation.verify().await.unwrap(), expected); + + let operation = replica + .small_root_origin_verification(&root, 64) + .await + .unwrap() + .unwrap(); + let packed = expected + .iter() + .find(|o| o.kind == CellObjectKind::Packed) + .unwrap(); + let path = + layout.incarnation_object_path(&root.cell, &root.incarnation, &packed.digest, packed.kind); + let bytes = backend.get(&path).await.unwrap().bytes().await.unwrap(); + let mut corrupt = bytes.to_vec(); + corrupt[64] ^= 1; + backend + .put(&path, Bytes::from(corrupt).into()) + .await + .unwrap(); + assert!(matches!( + operation.verify().await, + Err(cellule_ltx::LtxError::ChecksumMismatch) + )); + backend.put(&path, bytes.into()).await.unwrap(); + let operation = replica + .small_root_origin_verification(&root, 64) + .await + .unwrap() + .unwrap(); + backend.delete(&path).await.unwrap(); + assert!( + operation.verify().await.is_err(), + "prior inventory does not establish fresh body presence" + ); + assert!(matches!( + replica.small_root_origin_verification(&root, 0).await, + Err(cellule_ltx::LtxError::Limit( + cellule_ltx::LimitKind::RootInventoryObjects + )) + )); +} + +#[tokio::test] +async fn large_origin_graph_defers_to_the_canonical_streaming_inventory_and_exact_restore() { + let temp = tempfile::tempdir().unwrap(); + let mut db = Db::open(&temp.path().join("source"), Limits::default()).unwrap(); + let mut payload = vec![0_u8; 512 << 10]; + blake3::Hasher::new().finalize_xof().fill(&mut payload); + db.transaction(|tx| { + tx.execute_batch("CREATE TABLE t(v BLOB)")?; + tx.execute("INSERT INTO t VALUES(?1)", [&payload])?; + Ok(()) + }) + .unwrap(); + let cuts = db.capture().unwrap(); + let replica = replica(Store::new(Arc::new(InMemory::new())), [163; 32], [164; 16]); + let root = replica.prepare(None, &cuts, 1, 1).await.unwrap().root(); + assert!( + replica + .small_root_origin_verification(&root, 64) + .await + .unwrap() + .is_none() + ); + let objects = replica.reachable_objects_bounded(&root, 64).await.unwrap(); + assert!( + objects + .iter() + .any(|object| object.kind == CellObjectKind::Ltx) + ); + let destination = temp.path().join("cold"); + replica + .open_root(&root) + .await + .unwrap() + .restore(&destination) + .await + .unwrap(); + let conn = cellule_ltx::rusqlite::Connection::open(destination).unwrap(); + let restored: Vec = conn + .query_row("SELECT v FROM t", [], |row| row.get(0)) + .unwrap(); + assert_eq!(restored, payload); +} diff --git a/crates/cellule-runtime/src/node/bundle/proof.rs b/crates/cellule-runtime/src/node/bundle/proof.rs index 3852ffd5..4eac7752 100644 --- a/crates/cellule-runtime/src/node/bundle/proof.rs +++ b/crates/cellule-runtime/src/node/bundle/proof.rs @@ -274,6 +274,31 @@ pub(super) async fn verify_base( binding: &Binding, limits: cellule_ltx::Limits, ) -> Result<()> { + let (replica, base) = base_replica(layout, binding, limits)?; + // Reconstructability requires origin dependencies, even if metadata was + // authenticated earlier in this process. A cached root is not availability. + replica + .reachable_objects_bounded(&base, MAX_BASE_OBJECTS) + .await?; + Ok(()) +} + +pub(super) async fn prepare_base_origin( + layout: &cellule_ltx::CellStorageLayout, + binding: &Binding, + limits: cellule_ltx::Limits, +) -> Result> { + let (replica, base) = base_replica(layout, binding, limits)?; + Ok(replica + .small_root_origin_verification(&base, MAX_BASE_OBJECTS) + .await?) +} + +fn base_replica( + layout: &cellule_ltx::CellStorageLayout, + binding: &Binding, + limits: cellule_ltx::Limits, +) -> Result<(cellule_ltx::CellReplica, cellule_ltx::RootRef)> { let base = binding .control .ltx_root() @@ -284,12 +309,7 @@ pub(super) async fn verify_base( *binding.control.incarnation.as_bytes(), limits, )?; - // Reconstructability requires origin dependencies, even if metadata was - // authenticated earlier in this process. A cached root is not availability. - replica - .reachable_objects_bounded(&base, MAX_BASE_OBJECTS) - .await?; - Ok(()) + Ok((replica, base)) } impl BundleCoverageProof { diff --git a/crates/cellule-runtime/src/node/bundle/tests/faults.rs b/crates/cellule-runtime/src/node/bundle/tests/faults.rs index 4233cf1d..f57a749d 100644 --- a/crates/cellule-runtime/src/node/bundle/tests/faults.rs +++ b/crates/cellule-runtime/src/node/bundle/tests/faults.rs @@ -13,6 +13,7 @@ pub(super) struct ReplyFault { pub(super) node_updates: AtomicUsize, pub(super) coverage_puts: AtomicUsize, pub(super) range_started: AtomicUsize, + pub(super) base_started: AtomicUsize, pub(super) pin_started: tokio::sync::Notify, pub(super) pin_resume: tokio::sync::Notify, pub(super) node_started: tokio::sync::Notify, @@ -118,6 +119,14 @@ impl ObjectStore for ReplyFault { self.inner.put_multipart_opts(path, opts).await } async fn get_opts(&self, path: &Path, opts: GetOptions) -> object_store::Result { + let mode = self.mode.load(Ordering::SeqCst); + if (mode == 12 && path.as_ref().ends_with(".root")) + || (mode == 13 && path.as_ref().ends_with(".pack")) + { + self.base_started.fetch_add(1, Ordering::SeqCst); + self.node_started.notify_one(); + self.node_resume.notified().await; + } if self.mode.load(Ordering::SeqCst) == 11 && path.as_ref().ends_with(".cnb") && opts.range.is_some() diff --git a/crates/cellule-runtime/src/node/bundle/tests/index/base_cohort.rs b/crates/cellule-runtime/src/node/bundle/tests/index/base_cohort.rs new file mode 100644 index 00000000..1cbff0e5 --- /dev/null +++ b/crates/cellule-runtime/src/node/bundle/tests/index/base_cohort.rs @@ -0,0 +1,142 @@ +use super::*; +use std::sync::atomic::Ordering; + +#[tokio::test] +async fn fresh_base_root_reads_overlap_without_exceeding_the_original_admission() { + held_bases(12).await; +} + +#[tokio::test] +async fn fresh_base_body_reads_overlap_without_selecting_before_verification() { + held_bases(13).await; +} + +async fn held_bases(mode: u8) { + let faults = Arc::new(super::super::faults::ReplyFault::default()); + let mut f = Fixture::with_store(faults.clone()).await; + super::super::coverage::enroll(&mut f).await; + let mut cells = Vec::new(); + for number in 4..14 { + cells.push(f.cell(number).await); + f.heartbeat().await; + } + let mut frames = Vec::new(); + let mut assigned = Vec::new(); + for cell in &mut cells { + let (_, capture, range) = f.append(cell, 2); + frames.extend(capture); + assigned.push(range); + } + let now = f.node.advertisement().issued_at_ms(); + let proposal = f + .directory + .prepare_node_bundle(&f.node, &frames, &assigned, now) + .await + .unwrap(); + f.count.reset(); + faults.mode.store(mode, Ordering::SeqCst); + { + let selection = + f.directory + .select_node_bundle(&f.node, &proposal, &f.lease, Limits::default(), now); + tokio::pin!(selection); + let overlap = async { + loop { + let started = faults.node_started.notified(); + if faults.base_started.load(Ordering::SeqCst) >= 8 { + break; + } + started.await; + } + }; + tokio::select! { + _ = &mut selection => panic!("held base dependencies must prevent selection"), + result = tokio::time::timeout(std::time::Duration::from_secs(5), overlap) => result.unwrap(), + } + assert_eq!( + faults.base_started.load(Ordering::SeqCst), + 8, + "the next base cohort cannot exceed the admitted eight operations" + ); + } + assert_eq!( + f.count.put_requests(), + 0, + "cancelled bases select no authority" + ); + faults.mode.store(0, Ordering::SeqCst); + faults.node_resume.notify_waiters(); + f.count.reset(); + let (_, proofs) = f + .directory + .select_node_bundle(&f.node, &proposal, &f.lease, Limits::default(), now) + .await + .unwrap(); + assert_eq!(proofs.len(), cells.len()); + let roots = f + .count + .requests() + .into_iter() + .filter(|request| request.location.ends_with(".root")) + .count(); + assert_eq!(roots, cells.len(), "one fresh root body per base operation"); + for proof in &proofs { + assert_eq!(proof.commit_sequence(), 2); + let cell = cells + .iter() + .find(|cell| cell.control.value().bundle_binding == Some(proof.binding())) + .unwrap(); + let overlay = proof + .recovery_overlay(&f.layout, Limits::default()) + .await + .unwrap(); + let recovered = cell + .replica + .prepare_recovered_overlay(&overlay, 1) + .await + .unwrap(); + let destination = f.scratch.path().join(format!( + "base-cohort-{}.sqlite", + cell.control.value().cell.as_bytes()[0] + )); + cell.replica + .open_root(&recovered.root()) + .await + .unwrap() + .restore(&destination) + .await + .unwrap(); + let db = rusqlite::Connection::open(destination).unwrap(); + let outcomes: Vec<(String, String)> = db + .prepare("SELECT request, result FROM outcomes ORDER BY request") + .unwrap() + .query_map([], |row| Ok((row.get(0)?, row.get(1)?))) + .unwrap() + .collect::>() + .unwrap(); + assert_eq!( + outcomes, + vec![ + ("request-2".into(), "result-2".into()), + ("seed".into(), "original".into()) + ] + ); + } + // Prior successful verification does not establish current availability. + let base = cells[0].control.value().ltx_root().unwrap(); + let path = f.layout.incarnation_object_path( + &base.cell, + &base.incarnation, + &base.digest, + cellule_ltx::CellObjectKind::Root, + ); + f.count.block_body_reads_for(&path); + let puts = f.count.put_requests(); + assert!( + f.directory + .select_node_bundle(&f.node, &proposal, &f.lease, Limits::default(), now) + .await + .is_err() + ); + assert_eq!(f.count.put_requests(), puts); +} diff --git a/crates/cellule-runtime/src/node/bundle/tests/index/mod.rs b/crates/cellule-runtime/src/node/bundle/tests/index/mod.rs index 5826a9c0..82ddd578 100644 --- a/crates/cellule-runtime/src/node/bundle/tests/index/mod.rs +++ b/crates/cellule-runtime/src/node/bundle/tests/index/mod.rs @@ -38,6 +38,7 @@ fn head_session(f: &Fixture) -> SessionId { f.node.advertisement().session() } +mod base_cohort; mod bootstrap; mod checkpoint; mod cohort; diff --git a/crates/cellule-runtime/src/node/bundle/verification/base.rs b/crates/cellule-runtime/src/node/bundle/verification/base.rs new file mode 100644 index 00000000..52a3595c --- /dev/null +++ b/crates/cellule-runtime/src/node/bundle/verification/base.rs @@ -0,0 +1,50 @@ +//! Admitted fresh small-base verification; larger graphs stay serial. +use super::*; + +pub(super) async fn verify( + layout: &cellule_ltx::CellStorageLayout, + bindings: &[Binding], + limits: cellule_ltx::Limits, +) -> Result<()> { + for cohort in bindings.chunks(READ_CONCURRENCY) { + let mut small = Vec::with_capacity(cohort.len()); + let mut serial = Vec::with_capacity(cohort.len()); + let mut plans = stream::iter(0..cohort.len()) + .map(|index| async move { + Ok::<_, Error>(( + index, + proof::prepare_base_origin(layout, &cohort[index], limits).await?, + )) + }) + .buffer_unordered(READ_CONCURRENCY); + while let Some(plan) = plans.next().await { + let (index, plan): (_, Option) = plan?; + match plan { + Some(plan) => small.push(plan), + None => serial.push(index), + } + } + drop(plans); + let working = small.iter().try_fold(0_usize, |bytes, plan| { + bytes + .checked_add(plan.working_bytes()) + .ok_or(Error::Capacity("bundle base verification bytes")) + })?; + if working > MAX_BUNDLE_BYTES as usize { + return Err(Error::Capacity("bundle base verification bytes")); + } + let mut reads = stream::iter(small) + .map(|plan| async move { plan.verify().await.map(|_| ()) }) + .buffer_unordered(READ_CONCURRENCY); + while let Some(result) = reads.next().await { + result?; + } + drop(reads); + // All bounded operations have joined/dropped before a larger graph + // takes the original serial working set. No partial plan grants CAS. + for index in serial { + proof::verify_base(layout, &cohort[index], limits).await?; + } + } + Ok(()) +} diff --git a/crates/cellule-runtime/src/node/bundle/verification/mod.rs b/crates/cellule-runtime/src/node/bundle/verification/mod.rs index 98e11c96..30ed2ee5 100644 --- a/crates/cellule-runtime/src/node/bundle/verification/mod.rs +++ b/crates/cellule-runtime/src/node/bundle/verification/mod.rs @@ -6,6 +6,7 @@ use std::sync::Mutex; use tokio::sync::Semaphore; const READ_CONCURRENCY: usize = 8; +mod base; // Together scratch and metadata replace the released fresh-origin buffer; // they never add to the producer's original 20-MiB working reservation. const SCRATCH_BYTES: u64 = MAX_BUNDLE_BYTES / 2; @@ -142,10 +143,9 @@ pub(super) async fn verify_cohort( } return Ok(()); } - // Base traversal retains its original serial memory bound and fresh checks. - for binding in bindings { - proof::verify_base(layout, binding, limits).await?; - } + // Base and historical verification occupy the released origin allowance + // in disjoint phases. No historical facts or windows coexist with bases. + base::verify(layout, bindings, limits).await?; let windows = windows(bindings)?; let facts = Mutex::new( bindings diff --git a/scripts/perf/README.md b/scripts/perf/README.md index 1f7b818b..69722b09 100644 --- a/scripts/perf/README.md +++ b/scripts/perf/README.md @@ -1,12 +1,12 @@ # Docker write verification This harness runs the same SQL application on Cellule and pinned celld v0.6.1. -It uses 1,000 Cells, 96-byte values, INSERT plus SELECT in one transaction, +It defaults to 1,000 Cells, 96-byte values, INSERT plus SELECT in one transaction, and a two-hour durable request/result ledger. It is **SQL application parity**; it does not reproduce a bounded KV upsert benchmark. Use a dedicated Linux Docker context. The current shared-VM profile gives -nodes an 8-CPU/16-GiB ceiling, tmpfs state of 4 GiB, a 2-CPU/8-GiB RustFS +nodes an 8-CPU/16-GiB ceiling, tmpfs state of 4 GiB, a 2-CPU/2-GiB RustFS provider, and a 4-CPU/4-GiB client. These ceilings exceed the shared VM's total CPU; report contention. tmpfs does not qualify physical-device durability. HTTP endpoints, certificates, credentials, and placement are fixture policy. @@ -79,6 +79,9 @@ Its loopback ports must be free of other workloads. For a read-only point, supply `--write-rates 0 --read-rate N`. For a mixed point, use the same offered write rate on both systems and add `--read-rate N`. +Use `--cells 2000` for the node capacity contract; the population sets both the +Cellule application and client, and verifies celld's complete owner placement. +The simultaneous target is `--cells 2000 --write-rates 2000 --read-rate 20000`. Add `--hot-read-cells 10` for the 1% hot-read case. Record qualification of read-only and mixed profiles separately; an aggregate read/write rate is not a read-capacity result. diff --git a/scripts/perf/run.py b/scripts/perf/run.py index 43b2a0f3..0065164e 100644 --- a/scripts/perf/run.py +++ b/scripts/perf/run.py @@ -181,7 +181,7 @@ def start_node(system, durability, index, prefix, name): port = 8080 + index * 10 args = ['run', '-d', '--name', name, '--network', 'host', '--cpus', '8', '--memory', '16g', '--memory-swap', '16g', '--ulimit', 'nofile=65536:65536', '--tmpfs', '/scratch:rw,size=4g', *CREDS] if system == 'cellule': - args += ['-e', f'PARITY_RETAINED_BYTES={ARGS.retained_bytes}', '-e', f'PARITY_DISK_BYTES={ARGS.disk_bytes}', '-v', f'{BASE}:/work:ro', '-e', 'TMPDIR=/scratch', '-e', 'CELLULE_AXUM_CELLS=1000', '-e', 'CELLULE_AXUM_WORKERS=8', '-e', f'CELLULE_AXUM_BIND=127.0.0.1:{port}', '-e', 'CELLULE_TEST_ENDPOINT=http://127.0.0.1:9000', '-e', 'CELLULE_TEST_BUCKET=comparison', '-e', f'CELLULE_TEST_PREFIX={prefix}'] + args += ['-e', f'PARITY_RETAINED_BYTES={ARGS.retained_bytes}', '-e', f'PARITY_DISK_BYTES={ARGS.disk_bytes}', '-v', f'{BASE}:/work:ro', '-e', 'TMPDIR=/scratch', '-e', f'CELLULE_AXUM_CELLS={ARGS.cells}', '-e', 'CELLULE_AXUM_WORKERS=8', '-e', f'CELLULE_AXUM_BIND=127.0.0.1:{port}', '-e', 'CELLULE_TEST_ENDPOINT=http://127.0.0.1:9000', '-e', 'CELLULE_TEST_BUCKET=comparison', '-e', f'CELLULE_TEST_PREFIX={prefix}'] if durability == 'fleet' or index == 0: args += ['-e', 'CELLULE_AXUM_FLEET_DIR=/scratch/fleet'] if durability == 'bucket' and index == 0: @@ -231,7 +231,7 @@ def driver(directory, label, config): return result def config(directory, label, write_rate, read_rate, offset=0, seed=True, seconds=30): - return {'address': '127.0.0.1:8080', 'cells': 1000, 'concurrency': int(ARGS.concurrency), 'queue_capacity': int(ARGS.queue_capacity), 'write_rate': write_rate, 'read_rate': read_rate, 'warmup_seconds': ARGS.warmup if seed else 1, 'seconds': ARGS.seconds if seed else seconds, 'evidence_directory': '/tmp/comparison-evidence', 'seed_file': f'/work/{directory.relative_to(BASE)}/seeds.json' if seed else None, 'write_offset': offset, 'metrics_urls': [], 'metrics_tls_directory': None} + return {'address': '127.0.0.1:8080', 'cells': ARGS.cells, 'concurrency': int(ARGS.concurrency), 'queue_capacity': int(ARGS.queue_capacity), 'write_rate': write_rate, 'read_rate': read_rate, 'warmup_seconds': ARGS.warmup if seed else 1, 'seconds': ARGS.seconds if seed else seconds, 'evidence_directory': '/tmp/comparison-evidence', 'seed_file': f'/work/{directory.relative_to(BASE)}/seeds.json' if seed else None, 'write_offset': offset, 'metrics_urls': [], 'metrics_tls_directory': None} def contract(directory): now = int(time.time() * 1000) @@ -277,7 +277,7 @@ def run_case(system, durability): directory.mkdir(exist_ok=False) (directory / 'runner.py').write_bytes(RUNNER_BYTES) prefix = label + '-' + uuid.uuid4().hex[:12] - put(directory / 'case.json', {'system': system, 'durability': durability, 'prefix': prefix, 'framework_commit': MANIFEST['framework_revision'] if system == 'cellule' else 'f2bf648663a610eefde71f3547ad61e9b896b1f0', 'resident_cells': 1000, 'concurrency': int(ARGS.concurrency), 'queue_capacity': int(ARGS.queue_capacity), 'candidate_binary': MANIFEST['binaries']['sql']['path'], 'celld_application': 'celld-app', 'retained_budget_bytes': int(ARGS.retained_bytes), 'managed_disk_budget_bytes': int(ARGS.disk_bytes), 'value_bytes': 96, 'owner_cpus': 8, 'owner_memory_bytes': 16 * 1024 ** 3, 'followers': 2 if durability == 'fleet' else 0, 'profile': 'shared-vm-sql-ledger-96', 'provider_storage': 'fresh Linux Docker volume', 'provider_filesystem_required': True, 'provider_lifecycle_required': True, 'telemetry': ARGS.telemetry, 'warmup_seconds': ARGS.warmup, 'seconds': ARGS.seconds, 'runner_sha256': RUNNER_SHA256, 'docker_host': DOCKER_HOST, 'diagnostic': ARGS.seconds < 300 or ARGS.warmup < 30 or ARGS.telemetry == 'off'}) + put(directory / 'case.json', {'system': system, 'durability': durability, 'prefix': prefix, 'framework_commit': MANIFEST['framework_revision'] if system == 'cellule' else 'f2bf648663a610eefde71f3547ad61e9b896b1f0', 'resident_cells': ARGS.cells, 'concurrency': int(ARGS.concurrency), 'queue_capacity': int(ARGS.queue_capacity), 'candidate_binary': MANIFEST['binaries']['sql']['path'], 'celld_application': 'celld-app', 'retained_budget_bytes': int(ARGS.retained_bytes), 'managed_disk_budget_bytes': int(ARGS.disk_bytes), 'value_bytes': 96, 'owner_cpus': 8, 'owner_memory_bytes': 16 * 1024 ** 3, 'followers': 2 if durability == 'fleet' else 0, 'profile': 'shared-vm-sql-ledger-96', 'provider_storage': 'fresh Linux Docker volume', 'provider_filesystem_required': True, 'provider_lifecycle_required': True, 'telemetry': ARGS.telemetry, 'warmup_seconds': ARGS.warmup, 'seconds': ARGS.seconds, 'runner_sha256': RUNNER_SHA256, 'docker_host': DOCKER_HOST, 'diagnostic': ARGS.seconds < 300 or ARGS.warmup < 30 or ARGS.telemetry == 'off'}) names = [] stop = threading.Event() sampling = None @@ -339,8 +339,8 @@ def run_case(system, durability): if system == 'celld': states = {str(i): http('GET', '/state', port=8081 + i * 10) for i in range(3 if durability == 'fleet' else 1)} put(directory / 'placement-before.json', states) - if states['0']['body']['owned_cells'] != 1000 or any((states[str(i)]['body']['owned_cells'] for i in range(1, 3 if durability == 'fleet' else 1))): - raise RuntimeError('celld placement is not one owner of 1000 Cells') + if states['0']['body']['owned_cells'] != ARGS.cells or any((states[str(i)]['body']['owned_cells'] for i in range(1, 3 if durability == 'fleet' else 1))): + raise RuntimeError(f'celld placement is not one owner of {ARGS.cells} Cells') snapshot(directory, 'before', names) for n, rate in enumerate(ARGS.write_rates or [30, 2000 if durability == 'bucket' else 15000]): point = config(directory, f'write-{rate}', rate, ARGS.read_rate, offset=n * 100000000) @@ -486,6 +486,8 @@ def provision(): parser.add_argument('--write-rates', type=int, nargs='+') parser.add_argument('--read-rate', type=int, default=0) parser.add_argument('--hot-read-cells', type=int) + parser.add_argument('--cells', type=int, default=1000, + help='uniformly active Cell population for both applications') parser.add_argument('--telemetry', choices=['window', 'off'], default='window', help='off is an exporter-overhead diagnostic and cannot qualify') parser.add_argument('--concurrency', type=int, default=128) @@ -502,6 +504,8 @@ def provision(): parser.error('artifacts must stay outside the repository') if not 1 <= ARGS.repetitions <= 10 or not 1 <= ARGS.seconds <= 3600 or (not 1 <= ARGS.warmup <= 60): parser.error('invalid bounded duration/repetition') + if ARGS.cells <= 0: + parser.error('Cell population must be positive') if ARGS.overload_capacity is not None and ARGS.overload_capacity <= 0: parser.error('overload capacity must be positive') DOCKER = ['docker', '--context', CTX] From cf49ef5637c40cc081365706669a7e2bf83f0a00 Mon Sep 17 00:00:00 2001 From: forhappy Date: Fri, 9 Oct 2026 01:09:00 -0700 Subject: [PATCH 064/102] Deploy matching Cell population in performance fixtures --- scripts/perf/README.md | 4 +++- scripts/perf/run.py | 14 +++++++++++--- 2 files changed, 14 insertions(+), 4 deletions(-) diff --git a/scripts/perf/README.md b/scripts/perf/README.md index 69722b09..6534b1dc 100644 --- a/scripts/perf/README.md +++ b/scripts/perf/README.md @@ -80,7 +80,9 @@ Its loopback ports must be free of other workloads. For a read-only point, supply `--write-rates 0 --read-rate N`. For a mixed point, use the same offered write rate on both systems and add `--read-rate N`. Use `--cells 2000` for the node capacity contract; the population sets both the -Cellule application and client, and verifies celld's complete owner placement. +Cellule application and client, writes celld's per-case deployment configuration, +and verifies celld's complete owner placement. The original build fixture stays +immutable; each celld case retains its deployed application and configuration. The simultaneous target is `--cells 2000 --write-rates 2000 --read-rate 20000`. Add `--hot-read-cells 10` for the 1% hot-read case. Record qualification of read-only and mixed profiles separately; an aggregate read/write rate is not a diff --git a/scripts/perf/run.py b/scripts/perf/run.py index 0065164e..45f8734b 100644 --- a/scripts/perf/run.py +++ b/scripts/perf/run.py @@ -301,7 +301,15 @@ def run_case(system, durability): ready = docker('exec', CONTROL, 'python3', '/work/wait-store-ready.py', timeout=70) put(directory / 'store-readiness.json', json.loads(ready.stdout)) if system == 'celld': - p = docker('run', '--rm', '--network', 'host', '--cpus', '2', '--memory', '2g', '--memory-swap', '2g', '--ulimit', 'nofile=65536:65536', '-v', f"{BASE}/{'celld-app'}:/app:ro", *CREDS, CELLD, 'deploy', '/app', '--bucket', f's3://comparison/{prefix}', '--endpoint', 'http://127.0.0.1:9000', '--region', 'us-east-1') + # Keep the build fixture immutable while recording the exact + # population deployed for this fresh case. + application = directory / 'celld-app' + shutil.copytree(BASE / 'celld-app', application) + deployment = application / 'wrangler.json' + settings = json.loads(deployment.read_text()) + settings['vars']['CELLS'] = str(ARGS.cells) + put(deployment, settings) + p = docker('run', '--rm', '--network', 'host', '--cpus', '2', '--memory', '2g', '--memory-swap', '2g', '--ulimit', 'nofile=65536:65536', '-v', f'{application}:/app:ro', *CREDS, CELLD, 'deploy', '/app', '--bucket', f's3://comparison/{prefix}', '--endpoint', 'http://127.0.0.1:9000', '--region', 'us-east-1') (directory / 'deploy.log').write_text(p.stdout + p.stderr) elif durability == 'fleet': for i in [1, 2]: @@ -504,8 +512,8 @@ def provision(): parser.error('artifacts must stay outside the repository') if not 1 <= ARGS.repetitions <= 10 or not 1 <= ARGS.seconds <= 3600 or (not 1 <= ARGS.warmup <= 60): parser.error('invalid bounded duration/repetition') - if ARGS.cells <= 0: - parser.error('Cell population must be positive') + if not 1 <= ARGS.cells <= 2000: + parser.error('Cell population must be between 1 and 2000') if ARGS.overload_capacity is not None and ARGS.overload_capacity <= 0: parser.error('overload capacity must be positive') DOCKER = ['docker', '--context', CTX] From 60c30936096f5e0e1b93dba11a2d6fe842ccc9bb Mon Sep 17 00:00:00 2001 From: forhappy Date: Fri, 9 Oct 2026 01:54:52 -0700 Subject: [PATCH 065/102] Record matched 2000-Cell write and read diagnostics --- .../docs/write-performance-design.md | 32 +-- docs/bundle-coverage-implementation.md | 20 +- docs/pr67-base-cohort-measurement.md | 187 ++++++++++++++++++ 3 files changed, 218 insertions(+), 21 deletions(-) create mode 100644 docs/pr67-base-cohort-measurement.md diff --git a/crates/cellule-runtime/docs/write-performance-design.md b/crates/cellule-runtime/docs/write-performance-design.md index 231d28ab..042f0dc7 100644 --- a/crates/cellule-runtime/docs/write-performance-design.md +++ b/crates/cellule-runtime/docs/write-performance-design.md @@ -1,15 +1,18 @@ # Node write and read performance design -The [bounded historical-read comparison](../../../docs/pr67-bounded-history-measurement.md) -now groups fresh ranges within the original 20-MiB admission. One original-profile -Fleet pair completes 307.60 writes/s versus 144.50 before, with worse successful -p99 latency. A matched 1-GiB control reaches 237.67 versus 181.45/s, with zero -request errors. All Cellule ACKs pass warm/cold audit. These short overloaded -observations remain unqualified; PR #67 is still a draft. +The [2,000-Cell comparison](../../../docs/pr67-base-cohort-measurement.md) +overlaps fresh small-base verification within the original 20-MiB admission. +One Fleet pair completes 372.87 writes/s versus 271.00 before; celld completes +1,999.80/s at 2,000 offered/s. Successful scheduled p99 improves to 932.77 ms, +but Cellule errors, drops and joined drain fail. All warm ACKs pass; cold audit +is not reached. Read-only throughput is 17,456.38/s versus celld's 19,973.25/s +at 20,000 offered/s, with drops in both. These short observations remain +unqualified; PR #67 is still a draft. The [submission diagnosis](../../../docs/pr67-submission-timing-measurement.md) -identified publication capacity held under the global issuance lock. That -coupling still queues native progress before follower proof starts. Repeated +identified publication capacity held under the global issuance lock. The latest +candidate still spends about 99% of successful submission time waiting for that +lock. This coupling queues native progress before follower proof starts. Repeated historical/base verification and sparse root checkpoints keep the publication consumer expensive. Matching celld requires reducing that work and separating native progress from bounded recoverable publication debt; larger queues alone @@ -83,10 +86,15 @@ cache. Historical objects and every Cell base dependency still require origin verification. After the fresh body matches, selection shares the proposal allocation and uses its released buffer allowance for 2 MiB of historical scratch and at most 2 MiB of compact facts/planning metadata. Eight bounded reads overlap; -every frame and original Cell chain remains canonically checked. Base traversal -and individual extents above 2 MiB retain serial verification. Selection retains -no reconstruction bodies beyond the operation. Working admission remains 20 MiB; -workload retention and protocol bounds remain unchanged. +every frame and original Cell chain remains canonically checked. Fresh small +packed leaf bases now overlap in a separate phase: eight one-use canonical +origin plans charge at most 512 KiB each, reusing the same 4-MiB allowance after +the complete cohort body matches. All base jobs join or drop before historical +scratch/facts admission. Larger base graphs and individual historical extents +above 2 MiB retain serial verification. Selection retains no reconstruction +bodies beyond the operation. Working admission remains 20 MiB; workload +retention and protocol bounds remain unchanged. This is a structural charge, +not allocator-profile qualification. The first end-to-end Fleet diagnostic of this connection failed throughput, availability and drain. It is experimental, not performance qualification. diff --git a/docs/bundle-coverage-implementation.md b/docs/bundle-coverage-implementation.md index bb8cae6e..33ff5586 100644 --- a/docs/bundle-coverage-implementation.md +++ b/docs/bundle-coverage-implementation.md @@ -1,16 +1,18 @@ # Bundle coverage implementation -The [bounded historical-read comparison](pr67-bounded-history-measurement.md) -groups fresh ranges within the original 20-MiB admission. One original-profile -pair completes 307.60 writes/s versus 144.50 before, with worse successful p99. -A 1-GiB control with zero request errors reaches 237.67 versus 181.45/s. -Complete Cellule warm/cold ACK audits pass, but all capacity profiles remain -unqualified. +The [2,000-Cell comparison](pr67-base-cohort-measurement.md) overlaps fresh +small-base verification within the original 20-MiB admission. One Fleet write +pair completes 372.87 writes/s versus 271.00 before; celld completes 1,999.80/s +at 2,000 offered/s. Candidate successful scheduled p99 improves to 932.77 ms, +but errors, drops and joined drain fail. All warm ACKs pass; Cellule cold audit +is not reached. Read-only throughput is 17,456.38/s versus celld's 19,973.25/s +at 20,000 offered/s, with drops in both. All profiles remain unqualified. The [submission diagnosis](pr67-submission-timing-measurement.md) identifies -publication capacity held under the global issuance lock. The current selector -still gates native progress before follower proof. Celld pipelines that progress -independently of bucket publication. PR #67 remains a draft. +publication capacity held under the global issuance lock. The latest candidate +still spends about 99% of successful submission time waiting for that lock. +Celld pipelines native progress independently of bucket publication. Similar +components do not imply the same response path. PR #67 remains a draft. The installed original producer provides admitted shared receipts, independent root materialization and complete live-writer closure. The Fleet SQL example diff --git a/docs/pr67-base-cohort-measurement.md b/docs/pr67-base-cohort-measurement.md new file mode 100644 index 00000000..e70a3e78 --- /dev/null +++ b/docs/pr67-base-cohort-measurement.md @@ -0,0 +1,187 @@ +# PR 67: bounded base verification and the 2,000-Cell comparison + +**Publication still throttles native issuance.** One matched Fleet write pair +completes **372.87 writes/s after versus 271.00 before**; celld completes +**1,999.80/s at 2,000 offered/s**. The change overlaps fresh small-base +verification within the original memory admission. It improves this observation +by 37.59%, but Cellule still fails throughput, latency, availability and drain. +PR #67 remains a draft; performance parity is not delivered. + +## Workload and identities + +| Dimension | This diagnostic | +| --- | --- | +| Before / candidate framework | `90f099a295158c6effe51079d35db07eab204d17` / `29b94157c5915819c65b5ee80a345c2f32a41f3c` | +| Fixture population correction | `cf49ef5637c40cc081365706669a7e2bf83f0a00`; production Rust is identical to the measured candidate | +| celld | `f2bf648663a610eefde71f3547ad61e9b896b1f0` | +| Population | 2,000 uniformly active Cells, independently verified in configuration, placement and seed receipts | +| Command | SQL INSERT and SELECT, 96-byte value, two-hour durable request/result ledger | +| Durability | Fleet, one owner and two followers; local state and follower logs on tmpfs | +| Client | 128 clients and 128 queue slots; writes and reads measured separately | +| Timing | 30-second warmup, 60-second window; one before/after write pair and one celld write point | +| Offered load | 2,000 writes/s or 20,000 reads/s | +| Cellule limits | Original 20-MiB producer admission, 64-MiB retained work, 1-GiB managed disk | + +Each case uses a fresh namespace and object-store volume. The client, auditor, +fixtures, images and request settings match across the corrected cases. No +build or contributor suite overlaps their timed windows. Both managed write +paths use SQLite WAL NORMAL. + +The shared ARM64 Docker VM has **8 CPUs and 8 GiB total RAM**. Each serving-node +container has an 8-CPU/16-GiB ceiling, the provider a 2-CPU/8-GiB ceiling and the +client a 4-CPU/4-GiB ceiling. These ceilings exceed aggregate VM capacity. Only +Cellule has explicit internal retained/disk limits. This is a matched workload +diagnostic, not dedicated 8-vCPU/16-GiB capacity qualification, a KV overwrite +benchmark or physical-media durability evidence. + +The runner now accepts `--cells` and deploys that population in the celld +application configuration as well as the Cellule environment and client. The +first setup exposed the old celld template's fixed 1,000-Cell population before +any celld measurement began. That controller was stopped after its baseline +child completed; all five cases below use the same corrected runner. The +excluded initial baseline reached 301.22/s and also failed drain. Its evidence +is preserved; it is not substituted for the corrected pair's 271.00/s baseline. + +## Write results + +Latency columns contain **successful responses only**, including trailing +completions. Errors and dropped offers remain separate failures. + +| System | In-window writes/s | Scheduled p99 ms | Request p99 ms | Measured errors | Dropped offers | +| --- | ---: | ---: | ---: | ---: | ---: | +| Before | 271.00 | 2,184.86 | 874.13 | 55,715 | 47,769 | +| Candidate | 372.87 | 932.77 | 531.98 | 49,986 | 47,642 | +| celld | 1,999.80 | 13.38 | 12.22 | 0 | 0 | + +Before completes 16,260 writes in-window and 256 afterward; candidate completes +22,372 in-window and none afterward; celld completes 119,988 in-window and 12 +afterward. Successful scheduled p99 improves 57.31% and request p99 improves +39.14% in this single pair. These are observations, not repeatable capacity. +Candidate warmup errors worsen from 14,450 to 16,496; warmup drops change from +33,450 to 30,692. Celld has zero warmup errors or drops. Its offered rate limits +this point, so 1,999.80/s is not a maximum-throughput measurement. + +| System | Complete ACK cohort | Warm reads and original retries | Joined drain | Cold reads and original retries | +| --- | ---: | --- | --- | --- | +| Before | 30,617 | All pass | Fails 120-second deadline | Not reached | +| Candidate | 37,185 | All pass | Fails 120-second deadline | Not reached | +| celld | 182,001 | All pass | 16.63 s | All pass | + +ACK cohorts include seed, contract, warmup and trailing successful writes. +Cellule owner logs report fenced heartbeat/drain retries. The cause of that +shutdown failure is not yet proven; missing cold evidence does not establish +mutation loss. The complete issued range must still join before departure. + +## Read-only results + +There is no before read-only case, so this comparison establishes neither a +read improvement nor a read-regression guardrail for the change. + +| System | In-window reads/s | Scheduled p50 / p95 / p99 ms | Request p99 ms | Errors | Dropped offers | +| --- | ---: | ---: | ---: | ---: | ---: | +| Candidate | 17,456.38 | 2.13 / 20.30 / 28.79 | 14.67 | 0 | 152,566 | +| celld | 19,973.25 | 1.48 / 2.66 / 3.43 | 2.19 | 0 | 1,573 | + +Candidate request p50/p95 are 0.70/10.73 ms; celld's are 0.73/1.42 ms. +Warmup drops are 413,815 and 786 respectively, with zero read errors. All +successful reads match the expected seed output and Cell/incarnation, with a +commit sequence at least the seed's. Both warm audits pass all 2,001 ACKs. +Cellule again fails the 120-second drain and does not reach cold audit; celld +drains in 3.54 seconds and passes all cold reads/retries. Failure even in this +read-only case warrants investigating closure scheduling and lease renewal +alongside dirty publication debt. + +All five reports fail qualification. Read/write load together, Bucket, three +five-minute repetitions, equal internal resource policies and the dedicated +standard-node environment remain unmeasured in this slice. + +## Why similar components have different throughput + +The [earlier exact submission diagnosis](pr67-submission-timing-measurement.md) +measured 727.30 ms waiting for the ordered lock and 6.14 ms waiting for a +publication slot while holding it. That serialized 6.15-ms service interval +permits about 163 submissions/s, consistent with its 158.37 HTTP writes/s. +The actual hot-path scheduling differs despite both systems using SQLite, LTX, +node logs and object storage. + +The current pair retains the same coupling: + +| Successful assignment phase | Before mean ms | Candidate mean ms | +| --- | ---: | ---: | +| Local load and validation | 0.10626 | 0.08800 | +| Global ordered-lock wait | 224.93167 | 166.24648 | +| Publication slot with lock held | 2.07012 | 1.55474 | +| Ticket assignment and enqueue | 0.00660 | 0.00698 | +| Complete submission | 227.11516 | 167.89669 | + +All phase counts reconcile to the 16,324/22,372 successful assignment callbacks, +and their partition residual is zero. This is a submission cohort, not a sum of +HTTP, SQL, follower or publication timers. Ordered-lock waiting remains about +99% of submission time. Candidate mean SQL-worker, capture and follower-proof +timers are 0.16, 0.14 and 10.44 ms in their respective cohorts. Those cohorts +overlap and the timers are not additive to submission time. + +[`assign_capture`](../crates/cellule-runtime/src/node/log_shipper/mod.rs) +holds the node-wide ordering mutex while reserving publication capacity, before +assigning and enqueueing frames. Slow verification or materialization therefore +throttles writers across otherwise independent Cells before follower proof. +The reservation prevents sequence gaps and bounds original publication debt; +moving the same wait outside the mutex alone does not speed the consumer. + +Celld's [Fleet loop](https://github.com/denoland/celld/blob/f2bf648663a610eefde71f3547ad61e9b896b1f0/crates/celld/ltx_repl.rs#L4514) +does not wait for bucket publication and pipelines rounds with ordered +completion. Its [follower stream](https://github.com/denoland/celld/blob/f2bf648663a610eefde71f3547ad61e9b896b1f0/crates/celld/node_log.rs#L160) +groups delivered append frames into one durable batch. Cellule currently awaits +each `append_batch` before starting the next shipping round. That is a separate +pipeline gap; publication-coupled admission is the measured dominant gap. + +Windowed storage GET/range attempts per completed write rise **6.73 to 7.14**; +successful PUTs fall **0.486 to 0.476**, and materialized commands/root rise +**6.97 to 9.37**. These counters include background work and exclude SDK retries. +Overlapping fresh reads reduces waiting; it does not eliminate repeated base, +historical, index and root I/O. The checkpoint cost target remains unmet. + +## Delivered change and verification + +`RootOriginVerification` is a one-use fresh origin plan. Small packed leaf graphs +with at most 32 inline descriptors charge 512 KiB per operation. Up to eight +operations overlap within 4 MiB. Fresh base verification and historical +scratch/facts use disjoint phases under the original **20-MiB** reservation. +Large graphs and oversized historical extents retain canonical serial fallback. +There is no cross-operation availability cache or new durability authority. +Missing/corrupt dependencies still prevent selection; ordinary canonical +scope, digest, graph, body, inventory and exact-range checks remain in force. +The bound is a structural working-set charge, not allocator-profile evidence. + +Two real selector regressions hold root or pack I/O, observe exactly eight +operations and no ninth, and verify cancellation selects no authority. Both +fail three times before and pass three times after. Fresh retry verifies all +ten proofs and exact cold SQLite outcomes. LTX cases cover corruption/deletion +after planning, zero objects and a large-graph serial fallback with exact restore. + +All contributor routes pass across immutable snapshots with identical production +Rust: **1,975 workspace tests pass, zero fail, 38 ignored; 60 local LTX tests +pass**. Format, all-target/all-feature check, Rust 1.97/1.99 Clippy, API docs, +boundaries/layout, Rust fences, document links, SQL/peer contracts and 32 harness +tests pass. The first compilation error and subsequent missing API inventory +entry are preserved alongside their successful repairs. All 1,310 Rust/Cargo +files match the verified snapshot and pinned Linux build. Independent journal +reconciliation confirms every offer, attempt, success, error, per-Cell count, +successful latency and complete ACK cohort for all five corrected cases. + +Next: diagnose and fix 2,000-Cell joined closure/lease progress; lower fresh +verification and checkpoint I/O; separate native issuance from bounded +recoverable publication debt; then add ordered follower pipelining. Preserve +exact suffix recovery and safe collection contracts throughout. Full read, +mixed, Bucket and lifecycle/performance qualification remains required. + +Raw sources, binaries, fixtures, failed attempts, journals and reconciliations +remain outside Git under +`/Volumes/Workspace/crabbuild-target/cellule-write-perf-8ad1/base-cohort-20261009-*`. +The two new baseline cases reuse the immutable `bounded-history-20261008-candidate` +build; the external evidence inventory records those new files separately and +rehashes the earlier frozen inventory. + +[Implementation](bundle-coverage-implementation.md) · +[Capacity contract](../crates/cellule-runtime/docs/write-performance-design.md) · +[Previous comparison](pr67-bounded-history-measurement.md). From d12eac881523cf00fcfdee9662e82e10c9627991 Mon Sep 17 00:00:00 2001 From: forhappy Date: Fri, 9 Oct 2026 02:24:05 -0700 Subject: [PATCH 066/102] Bound Cell closure callbacks to preserve node lease renewal --- .../docs/write-performance-design.md | 5 +- .../src/node/durability/mod.rs | 18 +- .../src/node/durability/tests/closure.rs | 244 ++++++++++++++++++ .../durability/{tests.rs => tests/mod.rs} | 2 + docs/bundle-coverage-implementation.md | 7 +- 5 files changed, 273 insertions(+), 3 deletions(-) create mode 100644 crates/cellule-runtime/src/node/durability/tests/closure.rs rename crates/cellule-runtime/src/node/durability/{tests.rs => tests/mod.rs} (99%) diff --git a/crates/cellule-runtime/docs/write-performance-design.md b/crates/cellule-runtime/docs/write-performance-design.md index 042f0dc7..081e0456 100644 --- a/crates/cellule-runtime/docs/write-performance-design.md +++ b/crates/cellule-runtime/docs/write-performance-design.md @@ -257,7 +257,10 @@ uses the original root path. A later cohort's proof can now be narrowed only by the original complete-capture assignment, including its exact descriptors and body digests. Materialized roots join the managed checkpoint callback before releasing their publisher; Cell close waits for the complete original issued -prefix, including prior Fleet ACKs. The producer's task and failure cause also +prefix, including prior Fleet ACKs. After prefix joining, an eight-callback +admission bounds the provider authority mutex queue ahead of lease renewal. +Fencing wakes outside waiters; cancellation returns callback admission without +reopening frozen issuance. The producer's task and failure cause also join at epoch shutdown. This provides fair selection/checkpoint turns, not a fair node materializer scheduler or the 215-command checkpoint density target. diff --git a/crates/cellule-runtime/src/node/durability/mod.rs b/crates/cellule-runtime/src/node/durability/mod.rs index 793cb73f..6c612066 100644 --- a/crates/cellule-runtime/src/node/durability/mod.rs +++ b/crates/cellule-runtime/src/node/durability/mod.rs @@ -2,7 +2,7 @@ use std::sync::Arc; use futures_util::future::BoxFuture; -use tokio::sync::{Mutex, OnceCell}; +use tokio::sync::{Mutex, OnceCell, Semaphore}; use crate::identity::NodeId; use crate::identity::SessionId; @@ -21,10 +21,14 @@ mod publication; pub use publication::{BundleCheckpoint, NodeBundlePublicationAuthority}; mod receipts; +const MAX_BUNDLE_CLOSE_CALLBACKS: usize = 8; + /// Original node authority used to enroll and close the Cells of an installed /// shared publication feed. Implementations serialize these mutations and shared /// selection with their original heartbeat/enrollment state. The host owns and /// joins the feed; the runtime still verifies Cell departure against origin. +/// Runtime close callbacks enter in cohorts of at most eight, after their +/// complete issued producer prefix joins, to bound the authority's renewal queue. pub trait NodeBundleAuthority: Send + Sync { /// Pins the Serving Cell before its SQL admission opens. fn bind<'a>( @@ -197,6 +201,7 @@ pub struct NodeDurability { closed: std::sync::atomic::AtomicBool, selection_resources: std::sync::OnceLock, bundle_authority: std::sync::OnceLock>, + bundle_closures: Semaphore, publisher: std::sync::OnceLock, } @@ -246,6 +251,7 @@ impl NodeDurability { closed: std::sync::atomic::AtomicBool::new(false), selection_resources: std::sync::OnceLock::new(), bundle_authority: std::sync::OnceLock::new(), + bundle_closures: Semaphore::new(MAX_BUNDLE_CLOSE_CALLBACKS), publisher: std::sync::OnceLock::new(), } } @@ -444,6 +450,16 @@ impl NodeDurability { if let Some(publisher) = self.publisher.get() { publisher.wait_through(issued.last_node_sequence()).await?; } + // Thousands of actors may close together. Bound callbacks queued on + // the provider's shared FIFO authority mutex so lease refresh does not + // wait behind the whole Cell population. Prefix joining precedes this + // admission: callbacks/checkpoints needed to drain it retain progress. + // Cancellation returns the permit, never reopens frozen issuance. + let _callback = tokio::select! { + result = self.bundle_closures.acquire() => result.map_err(|_| Error::RuntimeClosed)?, + () = self.node_lease.wait_fenced() => return Err(Error::Fenced), + }; + self.node_lease.check()?; bundle.close(authority, observed, issued).await?; self.node_lease.check() } diff --git a/crates/cellule-runtime/src/node/durability/tests/closure.rs b/crates/cellule-runtime/src/node/durability/tests/closure.rs new file mode 100644 index 00000000..44d9b896 --- /dev/null +++ b/crates/cellule-runtime/src/node/durability/tests/closure.rs @@ -0,0 +1,244 @@ +//! Exercise callback scheduling through the original complete Cell close path. +use super::*; +use crate::control::authority::{CellAuthority, VersionedControl}; +use crate::control::{BundleBindingRef, Control, ControlState, Owner, RootRef}; +use crate::identity::Digest; +use crate::node::log::CellIssuedRange; +use futures_util::{future::join_all, poll}; +use std::sync::atomic::{AtomicUsize, Ordering}; + +struct FifoClosures { + state: tokio::sync::Mutex<()>, + lease: NodeLeaseGuard, + entered: AtomicUsize, + peak: AtomicUsize, + closed: Mutex>, +} + +struct Callback<'a>(&'a AtomicUsize); +impl Drop for Callback<'_> { + fn drop(&mut self) { + self.0.fetch_sub(1, Ordering::SeqCst); + } +} + +impl NodeBundleAuthority for FifoClosures { + fn bind<'a>( + &'a self, + _: &'a CellAuthority, + _: &'a VersionedControl, + ) -> BoxFuture<'a, Result> { + Box::pin(async { Err(Error::Node("unused closure test bind")) }) + } + fn close<'a>( + &'a self, + authority: &'a CellAuthority, + observed: &'a VersionedControl, + issued: CellIssuedRange, + ) -> BoxFuture<'a, Result<()>> { + Box::pin(async move { + let entered = self.entered.fetch_add(1, Ordering::SeqCst) + 1; + self.peak.fetch_max(entered, Ordering::SeqCst); + let _callback = Callback(&self.entered); + let _state = self.state.lock().await; + self.lease.check()?; + let scope = issued.scope(); + assert_eq!( + scope.application.as_bytes(), + authority.layout().application_id() + ); + assert_eq!(scope.cell, observed.value().cell); + assert_eq!(scope.incarnation, observed.value().incarnation); + assert_eq!(scope.cell_epoch, observed.value().epoch); + assert_eq!(issued.leader_session(), session(1)); + assert_eq!(issued.log_epoch(), 2); + assert_eq!(issued.commit_sequence(), 1); + assert_eq!( + issued.position(), + observed.value().ltx_root().unwrap().position + ); + assert_eq!(issued.last_node_sequence(), 0); + self.closed.lock().unwrap().push(issued); + Ok(()) + }) + } +} + +async fn fixture( + count: usize, +) -> ( + Arc, + Arc, + Vec<(CellAuthority, VersionedControl)>, +) { + let guard = lease(); + let callbacks = Arc::new(FifoClosures { + state: tokio::sync::Mutex::new(()), + lease: guard.clone(), + entered: AtomicUsize::new(0), + peak: AtomicUsize::new(0), + closed: Mutex::new(Vec::new()), + }); + let gate = DurabilityGate::new(session(1), node(1), 2, [node(2)]).unwrap(); + let transport: Arc = Arc::new(ImmediateTransport); + let shipper = NodeLogShipper::new( + gate.clone(), + transport.clone(), + cellule_ltx::Limits::default(), + ) + .unwrap(); + let durability = Arc::new(NodeDurability::new( + gate, + shipper, + Arc::new(RecordingAuthority::default()), + transport, + guard, + )); + // This seam verifies scheduling, not catalog publication or reconstruction. + // Real close callbacks share this FIFO mutex with their heartbeat authority. + let _feed = durability + .enable_bundle_publication(callbacks.clone()) + .unwrap(); + let layout = cellule_ltx::CellStorageLayout::new( + cellule_store::Store::new(Arc::new(object_store::memory::InMemory::new())), + object_store::path::Path::from("close-scheduling"), + [1; 16], + ); + let authority = CellAuthority::new(layout.clone()); + let mut cells = Vec::with_capacity(count); + for index in 1..=count { + let mut id = [0; 32]; + id[..8].copy_from_slice(&(index as u64).to_le_bytes()); + let cell = CellId::from_bytes(id); + let mut control = Control::initial( + cell, + IncarnationId::from_bytes([3; 16]), + Owner { + session: session(1), + endpoint: "https://original.test".into(), + }, + Digest::from_bytes([5; 32]), + 1, + ) + .unwrap(); + control.state = ControlState::Serving; + control.root = Some(RootRef { + digest: Digest::from_bytes([7; 32]), + txid: 1, + checksum: cellule_ltx::types::CHECKSUM_FLAG | 1, + commit_sequence: 1, + }); + control.bundle_binding = Some(BundleBindingRef { + session: session(1), + epoch: 2, + digest: Digest::from_bytes(id), + }); + layout + .store() + .create_strict( + &layout.control_path(cell.as_bytes()), + Bytes::from(control.encode().unwrap()), + ) + .await + .unwrap(); + cells.push(( + authority.clone(), + authority.load(cell).await.unwrap().unwrap(), + )); + } + (durability, callbacks, cells) +} + +#[tokio::test] +async fn two_thousand_cell_closures_leave_a_bounded_heartbeat_queue() { + let (durability, callbacks, cells) = fixture(2000).await; + let held = callbacks.state.lock().await; + let mut closures: Vec<_> = cells + .iter() + .map(|(authority, observed)| Box::pin(durability.close_bundle_cell(authority, observed))) + .collect(); + // Poll every original caller once, including those outside the callback + // cohort. A join_all implementation may otherwise yield before all callers. + for close in &mut closures { + assert!(poll!(close.as_mut()).is_pending()); + } + let heartbeat = async { + let _state = callbacks.state.lock().await; + let before = callbacks.closed.lock().unwrap().len(); + callbacks.lease.renew(1001, 61001).unwrap(); + before + }; + tokio::pin!(heartbeat); + assert!(poll!(heartbeat.as_mut()).is_pending()); + drop(held); + let (results, before_heartbeat) = tokio::join!(join_all(closures), heartbeat); + assert!(results.into_iter().all(|result| result.is_ok())); + assert!( + before_heartbeat <= 8, + "heartbeat queued behind {before_heartbeat} Cell callbacks" + ); + assert!(callbacks.peak.load(Ordering::SeqCst) <= 8); + let closed = callbacks.closed.lock().unwrap(); + assert_eq!(closed.len(), 2000); + assert_eq!( + closed + .iter() + .map(|issued| issued.scope().cell) + .collect::>() + .len(), + 2000 + ); + assert_eq!(callbacks.entered.load(Ordering::SeqCst), 0); +} + +#[tokio::test] +async fn fencing_wakes_closures_waiting_outside_authority() { + let (durability, callbacks, cells) = fixture(16).await; + let held = callbacks.state.lock().await; + let mut closures: Vec<_> = cells + .iter() + .map(|(authority, observed)| Box::pin(durability.close_bundle_cell(authority, observed))) + .collect(); + for close in &mut closures { + assert!(poll!(close.as_mut()).is_pending()); + } + callbacks.lease.fence(); + for close in &mut closures[8..] { + assert!(matches!( + poll!(close.as_mut()), + std::task::Poll::Ready(Err(Error::Fenced)) + )); + } + closures.truncate(8); + drop(held); + assert!( + join_all(closures) + .await + .into_iter() + .all(|result| matches!(result, Err(Error::Fenced))) + ); + assert!(callbacks.closed.lock().unwrap().is_empty()); + assert_eq!(callbacks.entered.load(Ordering::SeqCst), 0); +} + +#[tokio::test] +async fn cancelling_queued_closures_releases_callback_admission() { + let (durability, callbacks, cells) = fixture(16).await; + let held = callbacks.state.lock().await; + let mut closures: Vec<_> = cells + .iter() + .map(|(authority, observed)| Box::pin(durability.close_bundle_cell(authority, observed))) + .collect(); + for close in &mut closures { + assert!(poll!(close.as_mut()).is_pending()); + } + drop(closures); + assert_eq!(callbacks.entered.load(Ordering::SeqCst), 0); + let mut retry = Box::pin(durability.close_bundle_cell(&cells[0].0, &cells[0].1)); + assert!(poll!(retry.as_mut()).is_pending()); + assert_eq!(callbacks.entered.load(Ordering::SeqCst), 1); + drop(held); + retry.await.unwrap(); + assert_eq!(callbacks.entered.load(Ordering::SeqCst), 0); + assert_eq!(callbacks.closed.lock().unwrap().len(), 1); +} diff --git a/crates/cellule-runtime/src/node/durability/tests.rs b/crates/cellule-runtime/src/node/durability/tests/mod.rs similarity index 99% rename from crates/cellule-runtime/src/node/durability/tests.rs rename to crates/cellule-runtime/src/node/durability/tests/mod.rs index 53903c02..bfec28d1 100644 --- a/crates/cellule-runtime/src/node/durability/tests.rs +++ b/crates/cellule-runtime/src/node/durability/tests/mod.rs @@ -11,6 +11,8 @@ use crate::node::log_transport::{ AppendRequest, NodeLogTransport, RetireRequest, SealRequest, TailRequest, }; +mod closure; + #[derive(Default)] struct AuthorityState { activations: Vec, diff --git a/docs/bundle-coverage-implementation.md b/docs/bundle-coverage-implementation.md index 33ff5586..a9e52baf 100644 --- a/docs/bundle-coverage-implementation.md +++ b/docs/bundle-coverage-implementation.md @@ -109,8 +109,13 @@ feed and admitted proof; the new connection is experimental and unqualified. The original SQL/capture/submission jobs must join before `close_cell_issuance`. Its ordered gate prevents late assignment from consuming a node sequence. The legacy identity-free `DurabilityGate::issue` cannot produce a per-Cell closure: -using it makes that closure fail closed. Managed close now waits for the complete +using it makes that closure fail closed. Managed close waits for the complete issued producer prefix and joins exact checkpoint callbacks before Cell departure. +After that prefix joins, at most eight close callbacks may enter the provider's +shared authority mutex queue. This bounds callbacks ahead of heartbeat renewal +during a population-wide drain. Lease fencing wakes callers outside the callback +cohort; cancellation returns admission while retaining frozen issuance. The +bound does not shorten an individual provider operation. Failed-actor closure and sustained materializer progress remain unqualified. ```mermaid From 61022523e6a3a12bda5c23fe936c1fa359d87eda Mon Sep 17 00:00:00 2001 From: forhappy Date: Fri, 9 Oct 2026 02:58:54 -0700 Subject: [PATCH 067/102] Record closure recovery measurements and preserve driver errors --- .../docs/write-performance-design.md | 19 +- docs/bundle-coverage-implementation.md | 17 +- docs/pr67-closure-admission-measurement.md | 172 ++++++++++++++++++ scripts/perf/run.py | 5 +- 4 files changed, 195 insertions(+), 18 deletions(-) create mode 100644 docs/pr67-closure-admission-measurement.md diff --git a/crates/cellule-runtime/docs/write-performance-design.md b/crates/cellule-runtime/docs/write-performance-design.md index 081e0456..483d4bef 100644 --- a/crates/cellule-runtime/docs/write-performance-design.md +++ b/crates/cellule-runtime/docs/write-performance-design.md @@ -1,17 +1,18 @@ # Node write and read performance design -The [2,000-Cell comparison](../../../docs/pr67-base-cohort-measurement.md) -overlaps fresh small-base verification within the original 20-MiB admission. -One Fleet pair completes 372.87 writes/s versus 271.00 before; celld completes -1,999.80/s at 2,000 offered/s. Successful scheduled p99 improves to 932.77 ms, -but Cellule errors, drops and joined drain fail. All warm ACKs pass; cold audit -is not reached. Read-only throughput is 17,456.38/s versus celld's 19,973.25/s -at 20,000 offered/s, with drops in both. These short observations remain -unqualified; PR #67 is still a draft. +The [current 2,000-Cell measurements](../../../docs/pr67-closure-admission-measurement.md) +complete joined drain and all-ACK warm/cold read and retry audits after bounding +provider closure callbacks. Fleet throughput is 339.30 writes/s and read-only +throughput is 16,410.03/s, both below the previous observations. Successful +scheduled write p99 is 1,824.77 ms; errors and drops remain. The first read setup +fails with HTTP 503 before any timed point; its unchanged retry does not erase +that failure. This is a closure scheduling fix, with no demonstrated throughput +gain. The previous celld reference completes 1,999.80 writes/s at 2,000 offered/s. +These short observations remain unqualified; PR #67 is still a draft. The [submission diagnosis](../../../docs/pr67-submission-timing-measurement.md) identified publication capacity held under the global issuance lock. The latest -candidate still spends about 99% of successful submission time waiting for that +candidate still spends 99.01% of successful submission time waiting for that lock. This coupling queues native progress before follower proof starts. Repeated historical/base verification and sparse root checkpoints keep the publication consumer expensive. Matching celld requires reducing that work and separating diff --git a/docs/bundle-coverage-implementation.md b/docs/bundle-coverage-implementation.md index a9e52baf..d9e5fe07 100644 --- a/docs/bundle-coverage-implementation.md +++ b/docs/bundle-coverage-implementation.md @@ -1,16 +1,17 @@ # Bundle coverage implementation -The [2,000-Cell comparison](pr67-base-cohort-measurement.md) overlaps fresh -small-base verification within the original 20-MiB admission. One Fleet write -pair completes 372.87 writes/s versus 271.00 before; celld completes 1,999.80/s -at 2,000 offered/s. Candidate successful scheduled p99 improves to 932.77 ms, -but errors, drops and joined drain fail. All warm ACKs pass; Cellule cold audit -is not reached. Read-only throughput is 17,456.38/s versus celld's 19,973.25/s -at 20,000 offered/s, with drops in both. All profiles remain unqualified. +The [current 2,000-Cell measurements](pr67-closure-admission-measurement.md) +complete joined drain and all-ACK warm/cold read and retry audits after bounding +provider closure callbacks. Fleet write throughput is 339.30/s and read-only +throughput is 16,410.03/s, both below the preceding observations. Successful +scheduled write p99 is 1,824.77 ms; errors and drops remain. The first read setup +fails with HTTP 503; its unchanged retry does not erase that failure. There is +no measured throughput improvement from this closure fix. All profiles remain +unqualified; the previous celld write reference is 1,999.80/s at 2,000 offered/s. The [submission diagnosis](pr67-submission-timing-measurement.md) identifies publication capacity held under the global issuance lock. The latest candidate -still spends about 99% of successful submission time waiting for that lock. +still spends 99.01% of successful submission time waiting for that lock. Celld pipelines native progress independently of bucket publication. Similar components do not imply the same response path. PR #67 remains a draft. diff --git a/docs/pr67-closure-admission-measurement.md b/docs/pr67-closure-admission-measurement.md new file mode 100644 index 00000000..44c16d98 --- /dev/null +++ b/docs/pr67-closure-admission-measurement.md @@ -0,0 +1,172 @@ +# PR 67: bounded closure admission and current performance + +**Joined drain and cold recovery now complete, but throughput parity remains +unmet.** Revision `d12eac881523cf00fcfdee9662e82e10c9627991` completes +339.30 Fleet writes/s and 16,410.03 read-only queries/s in separate short +diagnostics. Both points are lower than the previous observations; this change +delivers a closure scheduling fix, not a demonstrated throughput improvement. +The first read setup fails with HTTP 503 and remains an availability failure. +PR #67 remains a draft. + +## Reproduced shutdown failure and fix + +The previous [2,000-Cell comparison](pr67-base-cohort-measurement.md) fails +the unchanged 120-second joined-drain deadline, including in read-only load. +Sampled mutex timing in a separate instrumented read-only reproduction at +`60c30936096f5e0e1b93dba11a2d6fe842ccc9bb` establishes the scheduling failure: + +| Observation | Result | +| --- | ---: | +| Heartbeat wait on the example provider's shared FIFO mutex | 20,001 ms; lease expired before renewal | +| Largest sampled closure queue wait | 27,349 ms | +| Sampled completed closure mutex holds | 9–44 ms | +| Instrumented read-only drain | Fails the original 120-second deadline; cold audit not reached | + +Population-wide closure places too many callbacks ahead of heartbeat renewal. +The samples do not rule out an unsampled slow individual operation. Runtime +shutdown already joins before the example stops heartbeat renewal. + +`NodeDurability::close_bundle_cell` now admits at most eight provider closure +callbacks **after freezing issuance and joining the complete issued producer +prefix**. The lease is checked again before and after the original callback. +Fencing wakes callers waiting outside the callback cohort; cancellation returns +admission without reopening frozen issuance. No lease, drain deadline, capacity, +durability or persisted format is widened. + +The regression uses the real closure entry point for 2,000 bound controls and +a FIFO provider mutex shared with heartbeat. Two checks fail in each of three +before runs; all three pass in each of three after runs. They verify heartbeat +gets a turn, every exact scope closes, fenced queued callbacks do not enter, +and cancelled admission can be retried. Separate bundle tests retain real +selected-prefix, prior-Fleet-ACK, checkpoint and cold-recovery coverage. The +scheduling test's synthetic roots establish no physical durability claim. + +## Workload and identities + +| Dimension | This slice | +| --- | --- | +| Measured production revision | `d12eac881523cf00fcfdee9662e82e10c9627991`; diagnostic timing instrumentation removed | +| Population | 2,000 uniformly active Cells; all successful timed cases independently reconcile the full population | +| Command | SQL INSERT/SELECT, 96-byte value, two-hour durable request/result ledger | +| Durability | Fleet, one owner and two followers; local state and follower logs on tmpfs | +| Client and timing | 128 clients/queue slots; 30-second warmup, 60-second window; write-only and read-only separately | +| Offered load | 2,000 writes/s or 20,000 reads/s | +| Cellule limits | Original 20-MiB producer admission, 64-MiB retained work and 1-GiB managed disk | +| Previous celld reference | `f2bf648663a610eefde71f3547ad61e9b896b1f0`; no fresh celld point in this slice | + +Every attempt uses a fresh namespace and provider volume. Candidate cases use +the same immutable corrected runner, clients, auditor, fixture settings and +images as the previous comparison. Builds and contributor suites finish before +candidate timed windows. Both managed write paths already use SQLite WAL NORMAL. + +The shared ARM64 Docker VM has **8 CPUs and 8 GiB total RAM**. Serving containers +have 8-CPU/16-GiB ceilings, the provider 2-CPU/8-GiB and the client 4-CPU/4-GiB; +their ceilings exceed aggregate capacity. Internal resource policies remain +asymmetric. These are workload diagnostics, not dedicated standard-node capacity, +KV-overwrite or physical-media durability qualification. + +## Actual write and read results + +Latency covers successful responses, including trailing completions. Errors +and dropped offers remain failures. Previous points below are historical +references, not a newly paired control for closure admission. + +| Workload / revision | In-window successes/s | Scheduled p99 ms | Request p99 ms | Errors | Dropped offers | +| --- | ---: | ---: | ---: | ---: | ---: | +| Fleet write, previous Cellule | 372.87 | 932.77 | 531.98 | 49,986 | 47,642 | +| Fleet write, current Cellule | 339.30 | 1,824.77 | 657.85 | 52,299 | 47,216 | +| Fleet write, previous celld | 1,999.80 | 13.38 | 12.22 | 0 | 0 | +| Read-only, previous Cellule | 17,456.38 | 28.79 | 14.67 | 0 | 152,566 | +| Read-only, current Cellule retry | 16,410.03 | 30.95 | 15.42 | 0 | 215,374 | +| Read-only, previous celld | 19,973.25 | 3.43 | 2.19 | 0 | 1,573 | + +Current write completes 20,358 responses in-window and 127 afterward. Warmup +has 15,750 errors and 30,444 dropped offers. Successful scheduled p50/p95 are +452.24/676.88 ms; request p50/p95 are 231.77/367.11 ms. The observed rate is +9.00% lower than the previous point; this unpaired short observation establishes +neither an improvement nor a repeatable regression caused by the closure change. + +The read retry completes 984,602 responses in-window and 24 afterward. Warmup +has zero read errors and 417,074 drops. Successful scheduled p50/p95 are +2.14/22.34 ms; request p50/p95 are 0.64/11.28 ms. Its rate is 5.99% below the +previous point. No repeatable read gain or regression guardrail is established. + +| Current case | Complete ACK cohort | Warm reads / original retries | Joined drain | Cold reads / original retries | +| --- | ---: | --- | ---: | --- | +| Fleet write | 36,292 | All pass | 52.63 s | All pass | +| Read-only retry | 2,001 | All pass | 41.72 s | All pass | + +Cohorts include seeds, contract checks, warmup and trailing successful writes. +Every ACK mutation and original retry result passes both audits, with zero +errors or changed incarnations. All 2,000 Cells have timed successes. Journal +counts, receipt scope and expected outputs reconcile independently. These two +cases complete lifecycle checks; their performance reports still fail. + +The first read-only attempt aborts **during initialization**, after 1,783 +successful seed responses, on HTTP 503 (`Cell is temporarily unavailable`). +It produces no timed TPS and reaches no cold audit. The owner exits normally +without OOM; the cause remains unproven. The unchanged fresh-namespace retry +does not erase this availability failure. The runner previously masked the +driver's 503 with a missing-summary error; it now retains the original exit +and source output plus copied partial journals. A replay at the actual driver +call seam fails before and passes after, including preservation of nonzero +drivers that have a real measurement summary. Measured cases retain the earlier +immutable runner; this diagnostic fix changes none of their results. + +## Why write throughput is still low + +In the current write window, all eight submission phase counts match 20,422 +successful assignments and their time partition residual is zero: + +| Submission phase | Mean ms | +| --- | ---: | +| Local load and validation | 0.11691 | +| Global ordered-lock wait | 181.20062 | +| Publication capacity, with lock held | 1.68201 | +| Ticket assignment and enqueue | 0.00700 | +| Complete submission | 183.00707 | + +Ordered-lock waiting is **99.01% of submission time**. The canonical +[`assign_capture`](../crates/cellule-runtime/src/node/log_shipper/mod.rs) +still reserves publication capacity under the node-wide ordering mutex before +frames enter follower shipping. Background selection and materialization can +therefore throttle otherwise independent Fleet writers. Mean SQL worker, +capture and follower-proof timers are 0.20, 0.15 and 11.09 ms in their respective +overlapping cohorts; they are not additive to this submission partition. + +Windowed GET/range attempts are 7.09 per completed write; successful PUTs are +0.495 and materialized commands per root are 8.43. These include background work +and exclude SDK retries, so they are not exact per-command costs. Between window +boundary samples, pending publications grow 1,444→2,493 and retained bytes +50,290,456→64,769,848 against a 67,108,864-byte limit. Unpublished node-log bytes +grow 10,650,379→29,629,212. Two endpoints do not establish a sustained debt slope. + +Celld's [Fleet loop](https://github.com/denoland/celld/blob/f2bf648663a610eefde71f3547ad61e9b896b1f0/crates/celld/ltx_repl.rs#L4514) +separates follower progress from bucket publication and pipelines ordered +rounds. Its [follower stream](https://github.com/denoland/celld/blob/f2bf648663a610eefde71f3547ad61e9b896b1f0/crates/celld/node_log.rs#L160) +groups delivered append frames before durable append. Cellule awaits each +shipping batch. The measured dominant gap remains publication-coupled admission, +followed by publication cost and the separate shipping pipeline gap. + +Next work is bounded overlap of authenticated catalog/history reads, lower +root/checkpoint cost, recoverably bounded native/publication separation and +ordered shipping. Initial availability, mixed load, Bucket, sustained debt, +safe collection, full fault lifecycle and three paired five-minute repetitions +remain required. Larger queues and a passing drain alone cannot qualify parity. + +## Verification and evidence + +All contributor routes pass in a frozen source snapshot: 1,978 workspace tests, +60 local LTX tests, Rust 1.97/1.99 Clippy with warnings denied, format, all-feature +checks, docs and boundary/layout/document/SQL-peer gates. The 38 ignored tests +retain their documented environment requirements. All 1,311 Rust/Cargo files +match that verified snapshot and the measured Linux build. The later changes +are this report, status text and the source-preserving Python diagnostic. + +Raw source, build identity, binaries, timing samples, failed setup, regression +replays, journals, metrics, audits and the rehash inventory remain outside Git +under `/Volumes/Workspace/crabbuild-target/cellule-write-perf-8ad1/closure-admission-20261009-*`. +The external inventory rehashes the preceding frozen comparison as well. + +[Implementation](bundle-coverage-implementation.md), +[capacity contract](../crates/cellule-runtime/docs/write-performance-design.md). diff --git a/scripts/perf/run.py b/scripts/perf/run.py index 45f8734b..27cd9d04 100644 --- a/scripts/perf/run.py +++ b/scripts/perf/run.py @@ -225,7 +225,10 @@ def driver(directory, label, config): if copy.returncode: raise RuntimeError(f'no capacity evidence: {copy.stderr}; {p.stdout} {p.stderr}') docker('exec', CONTROL, 'rm', '-rf', '/tmp/comparison-evidence') - result = json.loads((dest / 'summary.json').read_text()) + summary_path = dest / 'summary.json' + if not summary_path.is_file(): + raise RuntimeError(f'{label} capacity driver exited {p.returncode} without a summary: {p.stdout} {p.stderr}') + result = json.loads(summary_path.read_text()) result['driver_exit_code'] = p.returncode put(dest / 'summary.json', result) return result From 1502cff0e454b953de7f9b2cb0fd743ad582288d Mon Sep 17 00:00:00 2001 From: forhappy Date: Fri, 9 Oct 2026 03:29:48 -0700 Subject: [PATCH 068/102] Overlap bounded authenticated catalog and history reads --- .../docs/write-performance-design.md | 9 + .../src/node/bundle/index/io.rs | 180 ++++++++++++++---- .../src/node/bundle/tests/faults.rs | 21 ++ .../bundle/tests/index/metadata_cohort.rs | 172 +++++++++++++++++ .../src/node/bundle/tests/index/mod.rs | 1 + docs/bundle-coverage-implementation.md | 4 + 6 files changed, 348 insertions(+), 39 deletions(-) create mode 100644 crates/cellule-runtime/src/node/bundle/tests/index/metadata_cohort.rs diff --git a/crates/cellule-runtime/docs/write-performance-design.md b/crates/cellule-runtime/docs/write-performance-design.md index 483d4bef..4406a4f2 100644 --- a/crates/cellule-runtime/docs/write-performance-design.md +++ b/crates/cellule-runtime/docs/write-performance-design.md @@ -97,6 +97,15 @@ bodies beyond the operation. Working admission remains 20 MiB; workload retention and protocol bounds remain unchanged. This is a structural charge, not allocator-profile qualification. +Catalog shard reads now overlap at most eight raw bodies after the complete +selected-shard byte preflight. Requested detached histories use a separate +eight-read phase after the shard jobs join and their aggregate byte preflight +passes. Both phases retain the original 4-MiB selected-metadata bound and serial +authenticated decoding. Planning keeps at most 256 one-byte shard IDs and eight +history indices; unrelated histories remain authenticated references. This +changes scheduling rather than request count or proof policy. Application +throughput verification for this change is still pending. + The first end-to-end Fleet diagnostic of this connection failed throughput, availability and drain. It is experimental, not performance qualification. The subsequent [coverage-race measurement](../../../docs/pr67-coverage-race-measurement.md) diff --git a/crates/cellule-runtime/src/node/bundle/index/io.rs b/crates/cellule-runtime/src/node/bundle/index/io.rs index 5d5e0067..be8d44f8 100644 --- a/crates/cellule-runtime/src/node/bundle/index/io.rs +++ b/crates/cellule-runtime/src/node/bundle/index/io.rs @@ -1,5 +1,9 @@ //! Bounded origin point lookups and streaming maintenance inventory checks. use super::*; +use futures_util::{StreamExt, future::try_join_all, stream}; + +const READ_CONCURRENCY: usize = 8; +type DecodedRows = (Vec, BTreeMap<[u8; 32], history::History>); async fn load_root( layout: &cellule_ltx::CellStorageLayout, @@ -43,7 +47,17 @@ async fn load_rows( shard: &Shard, id: u8, origin: Option<&super::super::origin::OriginBundle>, -) -> Result<(Vec, BTreeMap<[u8; 32], history::History>)> { +) -> Result { + let bytes = read_rows(layout, root, shard, origin).await?; + decode_rows(root, shard, id, bytes) +} + +async fn read_rows( + layout: &cellule_ltx::CellStorageLayout, + root: &Root, + shard: &Shard, + origin: Option<&super::super::origin::OriginBundle>, +) -> Result { let extent = &shard.extent; let object = extent .object @@ -57,6 +71,22 @@ async fn load_rows( origin, ) .await?; + if bytes.len() as u64 != extent.bytes { + return Err(Error::Node("bundle catalog shard digest differs")); + } + Ok(bytes) +} + +fn decode_rows( + root: &Root, + shard: &Shard, + id: u8, + bytes: Bytes, +) -> Result { + let object = shard + .extent + .object + .ok_or(Error::Node("unresolved bundle catalog shard"))?; let (mut leaf, mut histories) = decode_leaf_with_histories(bytes, shard, id, root)?; for locator in leaf .bindings @@ -131,10 +161,37 @@ async fn load_inner( if wanted.is_some_and(|wanted| !wanted.contains(&id)) { continue; } - let (rows, leaf_histories) = match &root.shards[usize::from(id)] { - Some(shard) => load_rows(layout, &root, shard, id, origin).await?, - None => (Vec::new(), BTreeMap::new()), - }; + if root.shards[usize::from(id)].is_none() { + loaded.insert(id, Vec::new()); + } + } + // Aggregate encoded bytes were checked before opening any read. At most + // eight raw bodies coexist within that same bound; decode stays serial. + // A fixed-header-sized list retains at most 256 one-byte shard IDs and no + // decoded siblings. Owned indices keep this future Send at authority seams. + let mut shards = Vec::with_capacity(SHARDS); + for (id, shard) in root.shards.iter().enumerate() { + if shard.is_some() && !wanted.is_some_and(|wanted| !wanted.contains(&(id as u8))) { + shards.push(id as u8); + } + } + let mut reads = stream::iter(shards) + .map(|id| { + let root = &root; + async move { + let shard = root.shards[usize::from(id)] + .as_ref() + .ok_or(Error::Node("bundle shard plan is absent"))?; + Ok::<_, Error>((id, read_rows(layout, root, shard, origin).await?)) + } + }) + .buffered(READ_CONCURRENCY); + while let Some(result) = reads.next().await { + let (id, bytes) = result?; + let shard = root.shards[usize::from(id)] + .as_ref() + .ok_or(Error::Node("bundle shard plan is absent"))?; + let (rows, leaf_histories) = decode_rows(&root, shard, id, bytes)?; for (pin, history) in leaf_histories { if histories.insert(pin, history).is_some() { return Err(Error::Node("bundle inventory repeats a Cell pin")); @@ -143,6 +200,7 @@ async fn load_inner( bindings.extend(rows.iter().cloned()); loaded.insert(id, rows); } + drop(reads); bindings.sort_unstable_by_key(|binding| { binding .control @@ -168,40 +226,7 @@ async fn load_inner( .ok_or(Error::Capacity("bundle selected history bytes"))?; } } - for binding in &mut bindings { - if cells.is_some_and(|cells| { - !cells.contains(&( - *binding.application.as_bytes(), - *binding.control.cell.as_bytes(), - )) - }) { - continue; - } - let Some(history) = histories.get_mut(&history::pin(binding)?) else { - continue; - }; - let object = history - .extent - .object - .ok_or(Error::Node("unresolved bundle history"))?; - let bytes = super::super::origin::read_range( - layout, - session, - head.epoch, - object, - history.extent.offset..history.extent.offset + history.extent.bytes, - origin, - ) - .await?; - let mut locators = history::decode(&bytes, session, head.epoch, binding, history)?; - for locator in &mut locators { - if locator.object.is_none() { - locator.object = Some(object); - } - } - history.loaded = Some(locators.clone()); - binding.locators = locators; - } + hydrate_histories(layout, &root, cells, origin, &mut bindings, &mut histories).await?; // Snapshot after requested histories have loaded: hydration itself must // not rewrite a shard that the caller never changes. let mut hydrated = BTreeMap::>::new(); @@ -230,6 +255,83 @@ async fn load_inner( Ok(catalog) } +async fn hydrate_histories( + layout: &cellule_ltx::CellStorageLayout, + root: &Root, + cells: Option<&BTreeSet>, + origin: Option<&super::super::origin::OriginBundle>, + bindings: &mut [Binding], + histories: &mut BTreeMap<[u8; 32], history::History>, +) -> Result<()> { + let mut next = 0; + while next < bindings.len() { + // Only eight compact indices are planned. The already checked shard + // plus requested-history aggregate bounds every returned raw body; + // shard reads have joined before this phase reuses their allowance. + let mut targets = Vec::with_capacity(READ_CONCURRENCY); + while targets.len() < READ_CONCURRENCY && next < bindings.len() { + let binding = &bindings[next]; + if !cells.is_some_and(|cells| { + !cells.contains(&( + *binding.application.as_bytes(), + *binding.control.cell.as_bytes(), + )) + }) && histories.contains_key(&history::pin(binding)?) + { + targets.push(next); + } + next += 1; + } + let borrowed_bindings = &*bindings; + let borrowed_histories = &*histories; + let bodies = try_join_all((0..targets.len()).map(|offset| { + let index = targets[offset]; + async move { + let history = borrowed_histories + .get(&history::pin(&borrowed_bindings[index])?) + .ok_or(Error::Node("bundle history plan is absent"))?; + let object = history + .extent + .object + .ok_or(Error::Node("unresolved bundle history"))?; + let body = super::super::origin::read_range( + layout, + root.session, + root.epoch, + object, + history.extent.offset..history.extent.offset + history.extent.bytes, + origin, + ) + .await?; + if body.len() as u64 != history.extent.bytes { + return Err(Error::Node("bundle history digest differs")); + } + Ok(body) + } + })) + .await?; + for (index, bytes) in targets.into_iter().zip(bodies) { + let binding = &mut bindings[index]; + let history = histories + .get_mut(&history::pin(binding)?) + .ok_or(Error::Node("bundle history plan is absent"))?; + let object = history + .extent + .object + .ok_or(Error::Node("unresolved bundle history"))?; + let mut locators = history::decode(&bytes, root.session, root.epoch, binding, history)?; + for locator in &mut locators { + if locator.object.is_none() { + locator.object = Some(object); + } + } + history.loaded = Some(locators.clone()); + binding.locators = locators; + } + } + Ok(()) +} + /// A clean maintenance result covers every indexed binding. Only one bounded /// shard's decoded rows are retained at a time; the global duplicate-pin set is /// bounded by MAX_BINDINGS. No missing shard is interpreted as an empty shard. diff --git a/crates/cellule-runtime/src/node/bundle/tests/faults.rs b/crates/cellule-runtime/src/node/bundle/tests/faults.rs index f57a749d..5dd851a0 100644 --- a/crates/cellule-runtime/src/node/bundle/tests/faults.rs +++ b/crates/cellule-runtime/src/node/bundle/tests/faults.rs @@ -14,6 +14,7 @@ pub(super) struct ReplyFault { pub(super) coverage_puts: AtomicUsize, pub(super) range_started: AtomicUsize, pub(super) base_started: AtomicUsize, + pub(super) held_metadata: std::sync::Mutex>, pub(super) pin_started: tokio::sync::Notify, pub(super) pin_resume: tokio::sync::Notify, pub(super) node_started: tokio::sync::Notify, @@ -120,6 +121,26 @@ impl ObjectStore for ReplyFault { } async fn get_opts(&self, path: &Path, opts: GetOptions) -> object_store::Result { let mode = self.mode.load(Ordering::SeqCst); + let metadata = match &opts.range { + Some(object_store::GetRange::Bounded(range)) if path.as_ref().ends_with(".cnb") => { + (mode == 14 && range.start != 0) + || (mode == 15 + && self + .held_metadata + .lock() + .unwrap() + .iter() + .any(|(object, offset)| { + object == path.as_ref() && *offset == range.start + })) + } + _ => false, + }; + if metadata { + self.range_started.fetch_add(1, Ordering::SeqCst); + self.node_started.notify_one(); + self.node_resume.notified().await; + } if (mode == 12 && path.as_ref().ends_with(".root")) || (mode == 13 && path.as_ref().ends_with(".pack")) { diff --git a/crates/cellule-runtime/src/node/bundle/tests/index/metadata_cohort.rs b/crates/cellule-runtime/src/node/bundle/tests/index/metadata_cohort.rs new file mode 100644 index 00000000..f967aaf8 --- /dev/null +++ b/crates/cellule-runtime/src/node/bundle/tests/index/metadata_cohort.rs @@ -0,0 +1,172 @@ +use super::*; +use std::sync::atomic::Ordering; + +#[tokio::test] +async fn preparation_overlaps_catalog_shards_under_the_original_metadata_bound() { + held_metadata(14).await; +} + +#[tokio::test] +async fn preparation_overlaps_requested_histories_without_granting_cancelled_coverage() { + held_metadata(15).await; +} + +async fn held_metadata(mode: u8) { + let faults = Arc::new(super::super::faults::ReplyFault::default()); + let mut f = Fixture::with_store(faults.clone()).await; + super::super::coverage::enroll(&mut f).await; + let mut cells = Vec::new(); + for number in 4..20 { + cells.push(f.cell(number).await); + f.heartbeat().await; + } + let shards = cells + .iter() + .map(|cell| catalog_index::shard(&[9; 16], cell.control.value().cell.as_bytes())) + .collect::>(); + assert!(shards.len() > 8); + let mut frames = Vec::new(); + let mut assigned = Vec::new(); + for cell in &mut cells { + let (_, capture, range) = f.append(cell, 2); + frames.extend(capture); + assigned.push(range); + } + let now = f.node.advertisement().issued_at_ms(); + let first = f + .directory + .prepare_node_bundle(&f.node, &frames, &assigned, now) + .await + .unwrap(); + f.node = f + .directory + .select_node_bundle(&f.node, &first, &f.lease, Limits::default(), now) + .await + .unwrap() + .0; + let catalog = load_catalog(&f.layout, head_session(&f), first.head) + .await + .unwrap(); + for cell in &cells { + let pin = cell.control.value().bundle_binding.unwrap().digest; + let history = catalog_index::history_extent(&catalog, pin).unwrap(); + faults.held_metadata.lock().unwrap().push(( + f.layout + .node_coverage_bundle_path( + head_session(&f).as_bytes(), + first.head.epoch, + history.object.unwrap().as_bytes(), + ) + .to_string(), + history.offset, + )); + } + frames.clear(); + assigned.clear(); + for cell in &mut cells { + let (_, capture, range) = f.append(cell, 3); + frames.extend(capture); + assigned.push(range); + } + f.count.reset(); + faults.mode.store(mode, Ordering::SeqCst); + { + let preparation = f + .directory + .prepare_node_bundle(&f.node, &frames, &assigned, now); + tokio::pin!(preparation); + let overlap = async { + loop { + let started = faults.node_started.notified(); + if faults.range_started.load(Ordering::SeqCst) >= 8 { + break; + } + started.await; + } + }; + tokio::select! { + _ = &mut preparation => panic!("held metadata must prevent upload and selection"), + result = tokio::time::timeout(std::time::Duration::from_secs(5), overlap) => result.unwrap(), + } + assert_eq!( + faults.range_started.load(Ordering::SeqCst), + 8, + "no ninth metadata read may enter a held cohort" + ); + } + assert_eq!( + f.count.put_requests(), + 0, + "cancelled metadata grants no upload or authority" + ); + faults.mode.store(0, Ordering::SeqCst); + faults.node_resume.notify_waiters(); + let prepared = f + .directory + .prepare_node_bundle(&f.node, &frames, &assigned, now) + .await + .unwrap(); + let (_, proofs) = f + .directory + .select_node_bundle(&f.node, &prepared, &f.lease, Limits::default(), now) + .await + .unwrap(); + assert_eq!(proofs.len(), cells.len()); + for proof in &proofs { + assert_eq!(proof.commit_sequence(), 3); + let cell = cells + .iter() + .find(|cell| cell.control.value().bundle_binding == Some(proof.binding())) + .unwrap(); + let overlay = proof + .recovery_overlay(&f.layout, Limits::default()) + .await + .unwrap(); + let recovered = cell + .replica + .prepare_recovered_overlay(&overlay, 1) + .await + .unwrap(); + let cold = f.scratch.path().join(format!( + "metadata-{}.sqlite", + cell.control.value().cell.as_bytes()[0] + )); + cell.replica + .open_root(&recovered.root()) + .await + .unwrap() + .restore(&cold) + .await + .unwrap(); + let db = rusqlite::Connection::open(cold).unwrap(); + let outcomes: Vec<(String, String)> = db + .prepare("SELECT request, result FROM outcomes ORDER BY request") + .unwrap() + .query_map([], |row| Ok((row.get(0)?, row.get(1)?))) + .unwrap() + .collect::>() + .unwrap(); + assert_eq!( + outcomes, + vec![ + ("request-2".into(), "result-2".into()), + ("request-3".into(), "result-3".into()), + ("seed".into(), "original".into()) + ] + ); + } + let dependency = Path::from(faults.held_metadata.lock().unwrap()[0].0.clone()); + f.count.block_body_reads_for(&dependency); + let puts = f.count.put_requests(); + assert!( + f.directory + .prepare_node_bundle(&f.node, &frames, &assigned, now) + .await + .is_err() + ); + assert_eq!( + f.count.put_requests(), + puts, + "earlier verification cannot replace missing metadata" + ); +} diff --git a/crates/cellule-runtime/src/node/bundle/tests/index/mod.rs b/crates/cellule-runtime/src/node/bundle/tests/index/mod.rs index 82ddd578..26579b78 100644 --- a/crates/cellule-runtime/src/node/bundle/tests/index/mod.rs +++ b/crates/cellule-runtime/src/node/bundle/tests/index/mod.rs @@ -47,3 +47,4 @@ mod copy_on_write; mod density; mod history_cohort; mod inventory; +mod metadata_cohort; diff --git a/docs/bundle-coverage-implementation.md b/docs/bundle-coverage-implementation.md index d9e5fe07..9739c0db 100644 --- a/docs/bundle-coverage-implementation.md +++ b/docs/bundle-coverage-implementation.md @@ -159,6 +159,10 @@ references without walking a predecessor chain. A point lookup fetches one header, one shard and only the requested Cell's history, then its exact native frames. A shared shard keeps unrelated histories as authenticated references, including when those bodies are unavailable. +Multi-Cell lookups overlap up to eight shard reads and then eight requested +history reads, retaining the aggregate 4-MiB metadata preflight and serial +authentication/decoding. The two phases do not retain their raw bodies together; +unrequested histories remain references. Cancellation grants no publication. Selection verifies every participating Cell's origin and native suffix before CAS; point selection grants no sibling drain or collection authority. Complete maintenance inventory still verifies every shard, retaining one From 98a368acdf51c7cb7af1e06eb544f08ab5a55f68 Mon Sep 17 00:00:00 2001 From: forhappy Date: Fri, 9 Oct 2026 03:36:02 -0700 Subject: [PATCH 069/102] Format the catalog row decoder signature --- crates/cellule-runtime/src/node/bundle/index/io.rs | 7 +------ 1 file changed, 1 insertion(+), 6 deletions(-) diff --git a/crates/cellule-runtime/src/node/bundle/index/io.rs b/crates/cellule-runtime/src/node/bundle/index/io.rs index be8d44f8..2c98bf68 100644 --- a/crates/cellule-runtime/src/node/bundle/index/io.rs +++ b/crates/cellule-runtime/src/node/bundle/index/io.rs @@ -77,12 +77,7 @@ async fn read_rows( Ok(bytes) } -fn decode_rows( - root: &Root, - shard: &Shard, - id: u8, - bytes: Bytes, -) -> Result { +fn decode_rows(root: &Root, shard: &Shard, id: u8, bytes: Bytes) -> Result { let object = shard .extent .object From 7d52e8863568b82c61b70e6f1a74000e3b1c65fb Mon Sep 17 00:00:00 2001 From: forhappy Date: Fri, 9 Oct 2026 04:22:27 -0700 Subject: [PATCH 070/102] Record catalog overlap measurements and remaining bottlenecks --- .../docs/write-performance-design.md | 26 +-- docs/bundle-coverage-implementation.md | 19 +- docs/pr67-catalog-overlap-measurement.md | 178 ++++++++++++++++++ docs/write-performance-delivery.md | 11 +- 4 files changed, 212 insertions(+), 22 deletions(-) create mode 100644 docs/pr67-catalog-overlap-measurement.md diff --git a/crates/cellule-runtime/docs/write-performance-design.md b/crates/cellule-runtime/docs/write-performance-design.md index 4406a4f2..00813db5 100644 --- a/crates/cellule-runtime/docs/write-performance-design.md +++ b/crates/cellule-runtime/docs/write-performance-design.md @@ -1,18 +1,19 @@ # Node write and read performance design -The [current 2,000-Cell measurements](../../../docs/pr67-closure-admission-measurement.md) -complete joined drain and all-ACK warm/cold read and retry audits after bounding -provider closure callbacks. Fleet throughput is 339.30 writes/s and read-only -throughput is 16,410.03/s, both below the previous observations. Successful -scheduled write p99 is 1,824.77 ms; errors and drops remain. The first read setup -fails with HTTP 503 before any timed point; its unchanged retry does not erase -that failure. This is a closure scheduling fix, with no demonstrated throughput -gain. The previous celld reference completes 1,999.80 writes/s at 2,000 offered/s. -These short observations remain unqualified; PR #67 is still a draft. +The [current paired 2,000-Cell measurements](../../../docs/pr67-catalog-overlap-measurement.md) +complete joined drain and all-ACK warm/cold read and retry audits in all six +cases. Bounded metadata overlap completes 551.73 Fleet writes/s versus 211.37 +before and 1,999.88 for fresh celld. Successful scheduled write p99 is 566.39 ms; +errors, drops and growing publication debt remain. Read-only throughput is +16,631.43/s versus 18,778.07 before and 19,994.58 for celld. The adverse read +result fails the matched guardrail. This single short pair establishes no +repeatable gain or regression attributable to the change. Earlier failures remain +in their [original report](../../../docs/pr67-closure-admission-measurement.md). +All profiles remain unqualified; PR #67 is still a draft. The [submission diagnosis](../../../docs/pr67-submission-timing-measurement.md) identified publication capacity held under the global issuance lock. The latest -candidate still spends 99.01% of successful submission time waiting for that +candidate still spends 98.75% of successful submission time waiting for that lock. This coupling queues native progress before follower proof starts. Repeated historical/base verification and sparse root checkpoints keep the publication consumer expensive. Matching celld requires reducing that work and separating @@ -103,8 +104,9 @@ eight-read phase after the shard jobs join and their aggregate byte preflight passes. Both phases retain the original 4-MiB selected-metadata bound and serial authenticated decoding. Planning keeps at most 256 one-byte shard IDs and eight history indices; unrelated histories remain authenticated references. This -changes scheduling rather than request count or proof policy. Application -throughput verification for this change is still pending. +changes scheduling rather than request count or proof policy. Its +[fresh paired diagnostic](../../../docs/pr67-catalog-overlap-measurement.md) +records a higher write rate and lower read rate; performance qualification fails. The first end-to-end Fleet diagnostic of this connection failed throughput, availability and drain. It is experimental, not performance qualification. diff --git a/docs/bundle-coverage-implementation.md b/docs/bundle-coverage-implementation.md index 9739c0db..0fc7cbbc 100644 --- a/docs/bundle-coverage-implementation.md +++ b/docs/bundle-coverage-implementation.md @@ -1,17 +1,18 @@ # Bundle coverage implementation -The [current 2,000-Cell measurements](pr67-closure-admission-measurement.md) -complete joined drain and all-ACK warm/cold read and retry audits after bounding -provider closure callbacks. Fleet write throughput is 339.30/s and read-only -throughput is 16,410.03/s, both below the preceding observations. Successful -scheduled write p99 is 1,824.77 ms; errors and drops remain. The first read setup -fails with HTTP 503; its unchanged retry does not erase that failure. There is -no measured throughput improvement from this closure fix. All profiles remain -unqualified; the previous celld write reference is 1,999.80/s at 2,000 offered/s. +The [current paired 2,000-Cell measurements](pr67-catalog-overlap-measurement.md) +complete joined drain and all-ACK warm/cold read and retry audits in all six +cases. Bounded metadata overlap completes 551.73 Fleet writes/s versus 211.37 +before and 1,999.88 for fresh celld. Successful scheduled write p99 is 566.39 ms; +errors, drops and growing publication debt remain. Read-only throughput is +16,631.43/s versus 18,778.07 before and 19,994.58 for celld. The adverse read +result fails the matched guardrail. This single short pair establishes no +repeatable gain or regression; all profiles remain unqualified. Earlier failures +remain in their [original report](pr67-closure-admission-measurement.md). The [submission diagnosis](pr67-submission-timing-measurement.md) identifies publication capacity held under the global issuance lock. The latest candidate -still spends 99.01% of successful submission time waiting for that lock. +still spends 98.75% of successful submission time waiting for that lock. Celld pipelines native progress independently of bucket publication. Similar components do not imply the same response path. PR #67 remains a draft. diff --git a/docs/pr67-catalog-overlap-measurement.md b/docs/pr67-catalog-overlap-measurement.md new file mode 100644 index 00000000..040bd6f8 --- /dev/null +++ b/docs/pr67-catalog-overlap-measurement.md @@ -0,0 +1,178 @@ +# PR 67: bounded catalog reads and fresh paired measurements + +**Write throughput is higher in this pair, read throughput is lower, and +performance parity remains unmet.** The candidate completes 551.73 Fleet +writes/s versus 211.37 before and 1,999.88 for celld at 2,000 offered/s. +Read-only throughput is 16,631.43/s versus 18,778.07 before and 19,994.58 for +celld at 20,000 offered/s. Every case drains and passes complete ACK warm/cold +read and original-retry audits. Errors, drops and failed performance gates +remain. PR #67 is a draft. + +## Reproduced scheduling bottleneck and change + +Bundle preparation previously read selected catalog shards and requested +detached histories serially. Two regressions hold real metadata reads during +`prepare_node_bundle` for sixteen enrolled SQLite Cells. Both fail in each of +three before runs; both pass in each of three after runs. + +The canonical point-lookup path now overlaps at most **eight raw shard bodies** +after its existing aggregate byte preflight. Those jobs join before a separate +phase overlaps at most eight requested raw history bodies. Authentication and +decoding remain serial and canonical. The original selected-shard plus +requested-history **4-MiB bound**, 20-MiB producer admission, authority checks, +proof policy, persisted formats and maintenance inventory path remain intact. +Planning retains at most 256 one-byte shard IDs and eight history indices. +No cross-operation availability cache is introduced. + +The regressions enforce no ninth held read, no upload or authority after +cancellation, fresh successful retry and exact cold reconstruction of seed and +command outcomes. Previously verified metadata cannot authorize a later +preparation whose origin metadata is missing. This establishes bounded overlap +and preserved proof behavior; it does not establish application capacity. + +The ranked hypotheses preceding measurement were serial metadata I/O, +remaining base/root verification cost, and provider/admission saturation. +The positive write-rate difference supports the scheduling hypothesis. The +remaining lock waits and adverse read result keep the other costs unresolved. + +## Workload and identity + +| Dimension | This diagnostic | +| --- | --- | +| Before | `d12eac881523cf00fcfdee9662e82e10c9627991` | +| Candidate | `98a368acdf51c7cb7af1e06eb544f08ab5a55f68` | +| Fresh celld control | `f2bf648663a610eefde71f3547ad61e9b896b1f0` | +| Population | 2,000 uniformly active Cells; every timed case verifies all 2,000 | +| Command | 96-byte SQL INSERT/SELECT values, two-hour durable request/result ledger | +| Fleet | One owner, two followers; local state and follower logs on tmpfs | +| Arrival and timing | 128 clients/queue slots, 30-second warmup, one 60-second window | +| Offered load | 2,000 writes/s or 20,000 reads/s, in separate cases | +| Cellule admission | Original 64-MiB retained budget, 1-GiB managed disk, 20-MiB producer reservation | + +All six cases use fresh namespaces and provider volumes, the same immutable +runner, client/auditor binaries, fixture settings and images. Builds and +contributor suites finish before all timed windows. Both managed SQLite write +paths already use WAL NORMAL. The candidate has no diagnostic timing overlay. + +The shared ARM64 Docker VM has **8 CPUs and 8 GiB total RAM**. Owner and follower +ceilings are each 8 CPUs/16 GiB, provider 2 CPUs/8 GiB and client 4 CPUs/4 GiB; +aggregate ceilings exceed VM capacity. Internal policies remain asymmetric. +The host has 12 logical CPUs/32 GiB, another running 8-CPU/16-GiB Colima profile, +and 5,107 MiB of swap used at environment capture. No VM settings change within +this comparison. These observations do not isolate interference or qualify +the specified dedicated 8-vCPU/16-GiB serving node. + +## Actual rates, latency and failures + +TPS counts successful completions inside the window. Table latency covers +successful responses, including trailing completions; errors and drops remain +failures. The canonical delivery gate also retains all-attempt histograms. + +| Point | Successes/s | Scheduled p99 ms | Request p99 ms | Errors | Dropped offers | +| --- | ---: | ---: | ---: | ---: | ---: | +| Cellule write, before | 211.37 | 8,087.95 | 6,458.85 | 40,094 | 67,168 | +| Cellule write, candidate | 551.73 | 566.39 | 342.85 | 38,642 | 48,068 | +| celld write | 1,999.88 | 16.16 | 14.85 | 2 | 0 | +| Cellule read-only, before | 18,778.07 | 21.93 | 12.14 | 0 | 73,289 | +| Cellule read-only, candidate | 16,631.43 | 27.33 | 13.15 | 0 | 202,055 | +| celld read-only | 19,994.58 | 3.86 | 2.71 | 0 | 300 | + +Write rate is **2.61 times** the fresh before observation. Read rate is +**11.43% lower**. The canonical matched read guardrail fails: throughput ratio +0.886 and all-attempt scheduled-p99 ratio 1.245, outside its 0.90/1.20 bounds; +both Cellule arms also drop offers. A single short pair establishes neither a +repeatable gain nor a repeatable regression attributable to this change. +The previous observation from the same before binary was 339.30 writes/s; +that separately recorded run illustrates variance and is not this paired control. +These offered points do not establish either system's maximum capacity. + +All Cellule write errors preserve HTTP 503 (`Cell is temporarily unavailable`). +Celld's two timed and three warmup errors preserve HTTP 400 (`request identity +expired or not yet valid`); they are not omitted from qualification. Write +warmup errors/drops are 42,732/12,898 before, 14,801/24,555 candidate and 3/0 +celld. Read warmup drops are 406,792 before, 448,045 candidate and 4,407 celld, +with zero read errors. All six setups complete; the earlier initialization +failure remains in its [original report](pr67-closure-admission-measurement.md). + +| Case | Complete ACK cohort | Warm reads / original retries | Joined drain | Cold reads / original retries | +| --- | ---: | --- | ---: | --- | +| Cellule write, before | 19,109 | All pass | 54.28 s | All pass | +| Cellule write, candidate | 55,935 | All pass | 46.59 s | All pass | +| celld write | 181,996 | All pass | 17.55 s | All pass | +| Cellule read-only, before | 2,001 | All pass | 32.65 s | All pass | +| Cellule read-only, candidate | 2,001 | All pass | 37.27 s | All pass | +| celld read-only | 2,001 | All pass | 4.69 s | All pass | + +Cohorts include seeds, contract checks, warmup and trailing successful writes. +Journals, per-Cell distribution, receipt identity, expected outputs and ACK +counts reconcile independently. Every ACK mutation and original retry passes +both audits, with zero errors or changed incarnations. Lifecycle success does +not qualify the failing performance points. + +## Remaining architectural gap + +| Submission phase, mean ms | Before | Candidate | +| --- | ---: | ---: | +| Validation, native-byte and shipping-slot admission | 0.00053 | 0.00043 | +| Local load | 0.13732 | 0.09691 | +| Global ordered-lock wait | 367.45810 | 117.14341 | +| Publication capacity with lock held | 3.78531 | 1.37390 | +| Ticket assignment and enqueue | 0.00587 | 0.00728 | +| Complete submission | 371.38713 | 118.62193 | + +All eight phase counts reconcile with 12,682 before and 33,168 candidate +successful assignments; the partition residual is zero. The candidate still +spends **98.75%** of submission time waiting for the global ordering lock. +[`assign_capture`](../crates/cellule-runtime/src/node/log_shipper/mod.rs) +reserves publication capacity under that lock before follower shipping. Mean +SQL-worker, capture and follower-proof timers are 0.21, 0.14 and 20.98 ms in +their respective overlapping cohorts; they are not additive to this partition. + +Windowed GET/range attempts per completed write are 5.97 before and 7.01 +candidate; successful PUTs are 0.350 and 0.330. Materialized commands/root +are 4.45 and 13.07. These include background work and exclude SDK retries; +they are not exact command costs. Overlap changes scheduling, not an operation's +request count. A total request-amplification reduction is not demonstrated. + +Candidate pending publications grow 1,463→2,405 and retained bytes +63,881,005→66,341,791 against 67,108,864 available. Unpublished node-log bytes +grow 17,767,294→54,088,305; oldest unpublished age is 45.09→54.50 seconds. +Two endpoints do not establish a sustained debt slope. Read-only windows still +materialize seed debt: 897 roots before and 973 candidate. Their lower read +rate and host interference require controlled follow-up, not causal attribution +from this pair alone. + +Celld's [Fleet loop](https://github.com/denoland/celld/blob/f2bf648663a610eefde71f3547ad61e9b896b1f0/crates/celld/ltx_repl.rs#L4514) +pipelines ordered rounds independently of bucket publication. Its +[follower stream](https://github.com/denoland/celld/blob/f2bf648663a610eefde71f3547ad61e9b896b1f0/crates/celld/node_log.rs#L160) +groups already-delivered append frames before durable append. Cellule awaits +each append batch. Similar storage components still have different critical +paths; WAL NORMAL alone does not close this gap. + +Next work is bounded authenticated root/checkpoint verification, recoverably +bounded separation of native admission from publication debt, ordered follower +pipelining and diagnosis of read regression and overload availability. Bucket +adapter integration, mixed load, safe collection and full fault lifecycle remain +unfinished. Qualification still requires three matched five-minute repetitions, +zero errors/drops, Fleet p99 ≤50 ms, Bucket p99 ≤200 ms, bounded debt and every +ACK surviving cold recovery on the specified serving node. + +## Verification and retained evidence + +All contributor routes pass in the exact frozen candidate: **1,980 workspace +tests**, 60 local LTX tests, 94 bundle tests, both new regressions in three +repeats, Rust 1.97/1.99 Clippy with warnings denied, all-feature checks, format, +API docs and boundary/layout/document/SQL-peer gates. The 38 ignored tests +retain their documented environments. All **1,312 Rust/Cargo files** match +the measured Linux build and final production source. Later changes are this +report and status text. Initial compile/lint/build-identity failures are retained; +the first formatting-mismatched build is unused for measurement. + +Raw source, build identities, binaries, failures, journals, metrics, audits and +rehash inventories remain outside Git under +`/Volumes/Workspace/crabbuild-target/cellule-write-perf-8ad1/catalog-overlap-20261009-*`. +The inventory rehashes the preceding frozen comparisons as well. All six +canonical reports and their comparison remain explicitly unqualified. + +[Implementation](bundle-coverage-implementation.md), +[capacity contract](../crates/cellule-runtime/docs/write-performance-design.md). diff --git a/docs/write-performance-delivery.md b/docs/write-performance-delivery.md index c7719bf4..47b4cb76 100644 --- a/docs/write-performance-delivery.md +++ b/docs/write-performance-delivery.md @@ -6,7 +6,16 @@ comparison retains exact retry and cold-state audits. **Celld write parity has not been established.** The [proposal](write-performance-proposal.md) remains the acceptance contract; completing tests or a load run does not pass its gates. -The latest [selection-readiness comparison](pr67-selection-readiness-measurement.md) +The latest [paired 2,000-Cell comparison](pr67-catalog-overlap-measurement.md) +measures `98a368a` at **551.73 Fleet writes/s versus 211.37 before and 1,999.88 +for fresh celld**. Read-only throughput falls 18,778.07→16,631.43/s versus +celld's 19,994.58/s. All six cases pass joined drain and complete ACK warm/cold +read/retry audits. The candidate still returns 38,642 write errors, drops offers +and accumulates publication debt; the matched read guardrail fails. One short +pair establishes no repeatable improvement or regression. Performance parity +remains unmet and PR #67 stays a draft. Earlier failures below remain evidence. + +The earlier [selection-readiness comparison](pr67-selection-readiness-measurement.md) measures `6c909a6` at **184.77 Fleet writes/s and 248.68 Bucket writes/s**, versus 195.13 and 247.55 before. Its delayed-selection actor regression passes, but Fleet still returns 328,243 measured errors and fails 19,797 of 21,366 warm From 30cd960ff78a2be31fe84493ad0038b2ad4be7ad Mon Sep 17 00:00:00 2001 From: forhappy Date: Fri, 9 Oct 2026 05:34:48 -0700 Subject: [PATCH 071/102] Reduce repeated catalog scans during bundle encoding --- .../bundle/index/{dense.rs => dense/mod.rs} | 49 +++---- .../src/node/bundle/index/dense/native.rs | 70 ++++++++++ .../src/node/bundle/tests/index/encoding.rs | 121 ++++++++++++++++++ .../src/node/bundle/tests/index/mod.rs | 1 + 4 files changed, 208 insertions(+), 33 deletions(-) rename crates/cellule-runtime/src/node/bundle/index/{dense.rs => dense/mod.rs} (87%) create mode 100644 crates/cellule-runtime/src/node/bundle/index/dense/native.rs create mode 100644 crates/cellule-runtime/src/node/bundle/tests/index/encoding.rs diff --git a/crates/cellule-runtime/src/node/bundle/index/dense.rs b/crates/cellule-runtime/src/node/bundle/index/dense/mod.rs similarity index 87% rename from crates/cellule-runtime/src/node/bundle/index/dense.rs rename to crates/cellule-runtime/src/node/bundle/index/dense/mod.rs index 5af72df5..85c89b1c 100644 --- a/crates/cellule-runtime/src/node/bundle/index/dense.rs +++ b/crates/cellule-runtime/src/node/bundle/index/dense/mod.rs @@ -1,6 +1,8 @@ //! One immutable upload holds changed shards, native frames and small histories. use super::*; +mod native; + pub(super) fn encode( catalog: &mut Catalog, frames: &[cellule_ltx::VerifiedNodeFrame], @@ -104,42 +106,21 @@ pub(super) fn encode( .checked_add(bytes) .ok_or(Error::Capacity("bundle shard bytes"))?; } - let mut native = Vec::with_capacity(frames.len()); - for frame in frames { - let digest = Digest::from_bytes(frame.digest()); - let mut matches = 0; - for binding in &mut catalog.bindings { - let id = binding_shard(binding); - for locator in &mut binding.locators { - if locator.object.is_none() && locator.frame_digest == digest { - if !changed.contains(&id) { - return Err(Error::Node("native frame belongs to an unchanged shard")); - } - locator.offset = offset; - locator.bytes = frame.encoded().len() as u64; - matches += 1; - } - } - } - if matches != 1 { - return Err(Error::Node("bundle frame locator is not unique")); - } - native.push(Locator { - object: None, - offset, - bytes: frame.encoded().len() as u64, - frame_digest: digest, - }); - offset = offset - .checked_add(frame.encoded().len() as u64) - .ok_or(Error::Capacity("bundle bytes"))?; - } + let native = native::assign(catalog, frames, &changed, &mut offset)?; let mut history_bodies = Vec::new(); for pin in new { + // Catalog validation established strict pin order before mutation; + // assigning native extents changes no pin or control field. let binding = catalog .bindings - .iter() - .find(|binding| history::pin(binding).ok() == Some(pin)) + .binary_search_by_key(&Some(pin), |binding| { + binding + .control + .bundle_binding + .map(|pin| *pin.digest.as_bytes()) + }) + .ok() + .and_then(|index| catalog.bindings.get(index)) .ok_or(Error::Node("bundle history binding is absent"))?; let bytes = history::encode(catalog.session, catalog.epoch, binding)?; let reference = histories @@ -159,7 +140,9 @@ pub(super) fn encode( if offset > MAX_BUNDLE_BYTES { return Err(Error::Capacity("bundle bytes")); } - let groups = grouped(catalog); + // Detached leaf encoding replaces every nonempty locator array with its + // authenticated history extent. Native offset updates cannot change those + // compact rows, so retain the original groups instead of cloning them again. let mut bodies = Vec::new(); for id in &changed { if let Some(reference) = &mut shards[usize::from(*id)] { diff --git a/crates/cellule-runtime/src/node/bundle/index/dense/native.rs b/crates/cellule-runtime/src/node/bundle/index/dense/native.rs new file mode 100644 index 00000000..e1369901 --- /dev/null +++ b/crates/cellule-runtime/src/node/bundle/index/dense/native.rs @@ -0,0 +1,70 @@ +//! One bounded index maps new frame digests to their unique catalog locators. +use super::*; + +pub(super) fn assign( + catalog: &mut Catalog, + frames: &[cellule_ltx::VerifiedNodeFrame], + changed: &BTreeSet, + offset: &mut u64, +) -> Result> { + if frames.len() > MAX_FRAMES { + return Err(Error::Capacity("bundle frame count")); + } + if frames.is_empty() { + return Ok(Vec::new()); + } + // Four KiB of fixed stack planning replaces frame-by-frame full-catalog + // scans. There is no additional retained allocation or admission budget. + let mut ordered = [([0_u8; 32], 0_usize); MAX_FRAMES]; + for (index, frame) in frames.iter().enumerate() { + ordered[index] = (frame.digest(), index); + } + let ordered = &mut ordered[..frames.len()]; + ordered.sort_unstable_by_key(|(digest, _)| *digest); + if ordered.windows(2).any(|pair| pair[0].0 == pair[1].0) { + return Err(Error::Node("bundle frame locator is not unique")); + } + let mut locations = [None; MAX_FRAMES]; + for (binding_index, binding) in catalog.bindings.iter().enumerate() { + let mut shard_changed = None; + for (locator_index, locator) in binding.locators.iter().enumerate() { + if locator.object.is_some() { + continue; + } + let Ok(index) = + ordered.binary_search_by_key(locator.frame_digest.as_bytes(), |entry| entry.0) + else { + continue; + }; + if !*shard_changed.get_or_insert_with(|| changed.contains(&binding_shard(binding))) { + return Err(Error::Node("native frame belongs to an unchanged shard")); + } + let frame_index = ordered[index].1; + if locations[frame_index] + .replace((binding_index, locator_index)) + .is_some() + { + return Err(Error::Node("bundle frame locator is not unique")); + } + } + } + let mut native = Vec::with_capacity(frames.len()); + // Digest sorting is lookup only: persisted offsets follow the original + // issued frame order, including multiple frames in one complete capture. + for (index, frame) in frames.iter().enumerate() { + let (binding, locator) = + locations[index].ok_or(Error::Node("bundle frame locator is not unique"))?; + let locator = catalog + .bindings + .get_mut(binding) + .and_then(|binding| binding.locators.get_mut(locator)) + .ok_or(Error::Node("bundle frame locator is not unique"))?; + locator.offset = *offset; + locator.bytes = frame.encoded().len() as u64; + native.push(locator.clone()); + *offset = offset + .checked_add(locator.bytes) + .ok_or(Error::Capacity("bundle bytes"))?; + } + Ok(native) +} diff --git a/crates/cellule-runtime/src/node/bundle/tests/index/encoding.rs b/crates/cellule-runtime/src/node/bundle/tests/index/encoding.rs new file mode 100644 index 00000000..79397605 --- /dev/null +++ b/crates/cellule-runtime/src/node/bundle/tests/index/encoding.rs @@ -0,0 +1,121 @@ +use super::*; + +struct FixedClock; + +impl cellule_ltx::environment::Clock for FixedClock { + fn unix_millis(&self) -> i64 { + NOW + } + + fn file_age(&self, _: &std::path::Path) -> std::io::Result { + Ok(std::time::Duration::ZERO) + } +} + +#[tokio::test] +async fn dense_multi_cell_encoding_preserves_the_original_native_order_and_bytes() { + let mut f = Fixture::with_capture_host( + Arc::new(InMemory::new()), + cellule_ltx::Host::default().with_clock(Arc::new(FixedClock)), + ) + .await; + super::super::coverage::enroll(&mut f).await; + let mut cells = Vec::new(); + for number in 4..68 { + cells.push(f.cell(number).await); + f.heartbeat().await; + } + // Populate the real authenticated index with unrelated neighbors. Their + // original roots are deliberately unavailable: encoding may retain their + // references, but selecting this cohort must verify only its own bases. + inventory(&mut f, &cells[0], 1_937).await; + let original_sizes = [504_123, 511_931, 518_715, 525_755, 532_603, 539_643]; + let mut latest = Vec::new(); + for commit in 2..=7 { + let now = f.heartbeat().await; + let mut frames = Vec::new(); + let mut assignments = Vec::new(); + for cell in &mut cells { + let (_, capture, assignment) = f.append(cell, commit); + frames.extend(capture); + assignments.push(assignment); + } + let proposal = f + .directory + .prepare_node_bundle(&f.node, &frames, &assignments, now) + .await + .unwrap(); + assert_eq!(proposal.body.len(), original_sizes[(commit - 2) as usize]); + let mut next_offset = None; + for frame in &frames { + let locators: Vec<_> = proposal + .catalog + .bindings + .iter() + .flat_map(|binding| &binding.locators) + .filter(|locator| { + locator.object.is_none() && locator.frame_digest.as_bytes() == &frame.digest() + }) + .collect(); + assert_eq!(locators.len(), 1); + let locator = locators[0]; + assert_eq!(*next_offset.get_or_insert(locator.offset), locator.offset); + let end = locator.offset + locator.bytes; + assert_eq!( + &proposal.body[locator.offset as usize..end as usize], + frame.encoded().as_ref(), + ); + next_offset = Some(end); + } + let (selected, proofs) = f + .directory + .select_node_bundle(&f.node, &proposal, &f.lease, Limits::default(), now) + .await + .unwrap(); + f.node = selected; + assert_eq!(proofs.len(), cells.len()); + latest = proofs; + } + for proof in &latest { + assert_eq!(proof.commit_sequence(), 7); + assert_eq!(proof.locator_count(), 6); + let cell = cells + .iter() + .find(|cell| cell.control.value().bundle_binding == Some(proof.binding())) + .unwrap(); + let overlay = proof + .recovery_overlay(&f.layout, Limits::default()) + .await + .unwrap(); + let recovered = cell + .replica + .prepare_recovered_overlay(&overlay, 1) + .await + .unwrap(); + let cold = f.scratch.path().join(format!( + "encoded-{}.sqlite", + cell.control.value().cell.as_bytes()[0] + )); + cell.replica + .open_root(&recovered.root()) + .await + .unwrap() + .restore(&cold) + .await + .unwrap(); + let db = rusqlite::Connection::open(cold).unwrap(); + let outcomes: Vec<(String, String)> = db + .prepare("SELECT request, result FROM outcomes ORDER BY request") + .unwrap() + .query_map([], |row| Ok((row.get(0)?, row.get(1)?))) + .unwrap() + .collect::>() + .unwrap(); + let mut expected: Vec<_> = (2..=7) + .map(|commit| (format!("request-{commit}"), format!("result-{commit}"))) + .collect(); + expected.push(("seed".into(), "original".into())); + expected.sort(); + assert_eq!(outcomes, expected); + } +} diff --git a/crates/cellule-runtime/src/node/bundle/tests/index/mod.rs b/crates/cellule-runtime/src/node/bundle/tests/index/mod.rs index 26579b78..55ceb6fa 100644 --- a/crates/cellule-runtime/src/node/bundle/tests/index/mod.rs +++ b/crates/cellule-runtime/src/node/bundle/tests/index/mod.rs @@ -45,6 +45,7 @@ mod cohort; mod compatibility; mod copy_on_write; mod density; +mod encoding; mod history_cohort; mod inventory; mod metadata_cohort; From b048409374ac7cac7f2641d06c0142382bbed022 Mon Sep 17 00:00:00 2001 From: forhappy Date: Fri, 9 Oct 2026 06:22:57 -0700 Subject: [PATCH 072/102] Record encoder cost measurements and remaining parity gaps --- .../docs/write-performance-design.md | 32 ++- docs/bundle-coverage-implementation.md | 22 +- docs/pr67-encoder-cost-measurement.md | 205 ++++++++++++++++++ docs/write-performance-delivery.md | 19 +- 4 files changed, 249 insertions(+), 29 deletions(-) create mode 100644 docs/pr67-encoder-cost-measurement.md diff --git a/crates/cellule-runtime/docs/write-performance-design.md b/crates/cellule-runtime/docs/write-performance-design.md index 00813db5..2e6d660b 100644 --- a/crates/cellule-runtime/docs/write-performance-design.md +++ b/crates/cellule-runtime/docs/write-performance-design.md @@ -1,19 +1,20 @@ # Node write and read performance design -The [current paired 2,000-Cell measurements](../../../docs/pr67-catalog-overlap-measurement.md) -complete joined drain and all-ACK warm/cold read and retry audits in all six -cases. Bounded metadata overlap completes 551.73 Fleet writes/s versus 211.37 -before and 1,999.88 for fresh celld. Successful scheduled write p99 is 566.39 ms; -errors, drops and growing publication debt remain. Read-only throughput is -16,631.43/s versus 18,778.07 before and 19,994.58 for celld. The adverse read -result fails the matched guardrail. This single short pair establishes no -repeatable gain or regression attributable to the change. Earlier failures remain -in their [original report](../../../docs/pr67-closure-admission-measurement.md). -All profiles remain unqualified; PR #67 is still a draft. +The [current paired 2,000-Cell measurements](../../../docs/pr67-encoder-cost-measurement.md) +complete joined drain and all-ACK warm/cold read and original-retry audits in +all six cases. Reduced encoder scanning completes 550.52 Fleet writes/s versus +512.02 before and 1,994.57 for fresh celld. Successful scheduled write p99 is +866.06 ms versus 853.65 before; errors increase while dropped offers decrease. +Read-only throughput is 17,877.42/s versus 13,780.35 before and 19,522.95 for +celld. Both Cellule read arms drop offers, so the canonical read guardrail fails. +One short pair establishes no repeatable or attributable gain. Publication +cost and debt remain; all profiles are unqualified and PR #67 stays a draft. +The [preceding comparison](../../../docs/pr67-catalog-overlap-measurement.md) +and its failures remain evidence. The [submission diagnosis](../../../docs/pr67-submission-timing-measurement.md) identified publication capacity held under the global issuance lock. The latest -candidate still spends 98.75% of successful submission time waiting for that +candidate still spends 98.41% of successful submission time waiting for that lock. This coupling queues native progress before follower proof starts. Repeated historical/base verification and sparse root checkpoints keep the publication consumer expensive. Matching celld requires reducing that work and separating @@ -108,6 +109,15 @@ changes scheduling rather than request count or proof policy. Its [fresh paired diagnostic](../../../docs/pr67-catalog-overlap-measurement.md) records a higher write rate and lower read rate; performance qualification fails. +Dense encoding now maps the original at-most-64 frame digests to unique +catalog locations in one locator scan, then assigns extents in issued order. +Strictly ordered pins support history lookup without full binding scans; +detached leaf encoding reuses the original shard groups. Fixed planning uses +about 4 KiB of stack without new retained admission, format or proof changes. +The [encoder comparison](../../../docs/pr67-encoder-cost-measurement.md) +preserves exact bytes in the external old-encoder oracle, but establishes no +repeatable application improvement and fails performance qualification. + The first end-to-end Fleet diagnostic of this connection failed throughput, availability and drain. It is experimental, not performance qualification. The subsequent [coverage-race measurement](../../../docs/pr67-coverage-race-measurement.md) diff --git a/docs/bundle-coverage-implementation.md b/docs/bundle-coverage-implementation.md index 0fc7cbbc..72088a58 100644 --- a/docs/bundle-coverage-implementation.md +++ b/docs/bundle-coverage-implementation.md @@ -1,18 +1,20 @@ # Bundle coverage implementation -The [current paired 2,000-Cell measurements](pr67-catalog-overlap-measurement.md) -complete joined drain and all-ACK warm/cold read and retry audits in all six -cases. Bounded metadata overlap completes 551.73 Fleet writes/s versus 211.37 -before and 1,999.88 for fresh celld. Successful scheduled write p99 is 566.39 ms; -errors, drops and growing publication debt remain. Read-only throughput is -16,631.43/s versus 18,778.07 before and 19,994.58 for celld. The adverse read -result fails the matched guardrail. This single short pair establishes no -repeatable gain or regression; all profiles remain unqualified. Earlier failures -remain in their [original report](pr67-closure-admission-measurement.md). +The [current paired 2,000-Cell measurements](pr67-encoder-cost-measurement.md) +complete joined drain and all-ACK warm/cold read and original-retry audits in +all six cases. Reduced encoder scanning completes 550.52 Fleet writes/s versus +512.02 before and 1,994.57 for fresh celld. Successful scheduled write p99 is +866.06 ms versus 853.65 before; errors increase while dropped offers decrease. +Read-only throughput is 17,877.42/s versus 13,780.35 before and 19,522.95 for +celld. Both Cellule read arms drop offers, so the canonical read guardrail fails. +One short pair establishes no repeatable or attributable gain. Publication +cost and debt remain; all profiles are unqualified and PR #67 stays a draft. +The [preceding comparison](pr67-catalog-overlap-measurement.md) +and its failures remain evidence. The [submission diagnosis](pr67-submission-timing-measurement.md) identifies publication capacity held under the global issuance lock. The latest candidate -still spends 98.75% of successful submission time waiting for that lock. +still spends 98.41% of successful submission time waiting for that lock. Celld pipelines native progress independently of bucket publication. Similar components do not imply the same response path. PR #67 remains a draft. diff --git a/docs/pr67-encoder-cost-measurement.md b/docs/pr67-encoder-cost-measurement.md new file mode 100644 index 00000000..ab874872 --- /dev/null +++ b/docs/pr67-encoder-cost-measurement.md @@ -0,0 +1,205 @@ +# PR 67: catalog encoding cost and fresh paired measurements + +**The candidate completes 550.52 Fleet writes/s versus 512.02 before and +1,994.57 for celld. Performance parity remains unmet.** Successful scheduled +write p99 is 866.06 ms versus 853.65 before; returned errors increase while +dropped offers decrease. Read-only throughput is 17,877.42/s versus 13,780.35 +before and 19,522.95 for celld. All six cases drain and pass complete ACK +warm/cold read and original-retry audits. Every performance profile fails. +PR #67 remains a draft. + +## Change and compatibility evidence + +The dense catalog encoder previously scanned the complete loaded catalog for +every new frame, repeatedly computed binding shard IDs, searched every binding +for each new history, and cloned shard groups again after assigning native +offsets. The preceding timing-only diagnosis measured 11.63 ms mean encoding +inside 93.85 ms selection service. Encoding was one cost among larger catalog, +upload, base and historical verification costs. + +The encoder now uses a fixed digest-to-frame lookup for the original maximum +of 64 frames. It scans locators once, records their unique catalog locations, +and assigns extents in the original issued order. Two fixed stack arrays use +about 4 KiB; no additional retained admission or heap cache is introduced. +History binding lookup binary-searches the strictly ordered, already validated +pins. Detached leaf encoding substitutes authenticated history extents, so it +can reuse the original groups without cloning all bindings again. Empty-frame +checkpoint encodings bypass native lookup. + +Persisted bytes, formats, authority pins, exact verification, cohort limits, +origin freshness and the 20-MiB working reservation remain unchanged. The real +SQLite regression covers six 64-frame cohorts with 2,000 catalog bindings, +original encoded lengths and issued offsets, exact frame bytes, selected proofs +and cold restoration of all 64 participating Cells' seed and command outcomes. +It passes three repetitions. An external test-only copy of the old encoder +compares complete bytes and selected heads against the candidate on the same +live inputs: all 95 bundle tests pass, including 819 logged native encodings +with identical bytes and heads. Exploratory debug timings from that oracle +are not release or application performance evidence. + +## Matched workload and identities + +| Dimension | This diagnostic | +| --- | --- | +| Before binary revision | `98a368acdf51c7cb7af1e06eb544f08ab5a55f68`, the preceding production candidate | +| Candidate binary revision | `30cd960ff78a2be31fe84493ad0038b2ad4be7ad` | +| Fresh celld | `f2bf648663a610eefde71f3547ad61e9b896b1f0` | +| Population | 2,000 uniformly offered Cells; all 2,000 have successful responses in every timed case | +| Command | 96-byte SQL INSERT/SELECT values, two-hour durable request/result ledger | +| Fleet | One owner and two followers; local state and follower logs on tmpfs | +| Arrival | 128 clients/queue slots, 30-second warmup, one 60-second window | +| Offered points | Separate 2,000-write/s and 20,000-read/s cases | +| Admission | Original 64-MiB retained budget, 1-GiB managed disk, 20-MiB producer reservation | + +Namespaces and provider volumes are fresh. Runner, clients, auditor, images, +fixtures and resource settings match across all six cases. No build or +contributor suite overlaps timed windows. Both managed SQLite paths use WAL +NORMAL. The candidate has no timing instrumentation overlay. + +The before executable is a byte-identical copy of the preceding measured +production executable; it is measured again here, not rebuilt or represented +by its historical result. Candidate SQL SHA-256 is +`fa2673bc8d7d2cbe8205a063a58c520abe6a99f9c9c261aacb45257ff2cf7864`; +before is `6c401ce0c1aaaa7fb351caeedac4c7f4f37a03217577e504c588f577c97c504f`. +All 1,936 exported framework files match the committed candidate at build time. +The 1,315 Rust/Cargo inputs remain identical after these documentation updates. + +The shared ARM64 Docker VM has 8 CPUs and approximately 8 GiB total memory +across owner, followers, provider and client. Container ceilings exceed that +capacity; internal policies remain asymmetric. Another host VM remains active. +The [preceding environment capture](pr67-catalog-overlap-measurement.md) +records swap use and host limits. VM settings remain unchanged within this +pair. This does not qualify a dedicated 8-vCPU/16-GiB serving node or +physical-media durability. No Bucket or mixed-load result is claimed. + +## Actual throughput, latency and failures + +TPS counts successful completions inside the window. Successful scheduled p99 +includes trailing successful responses. All-attempt p99 also includes fast +failures; it cannot substitute for successful latency. Drops are separate. + +| Point | Successes/s | Successful scheduled p99 ms | All-attempt scheduled p99 ms | Errors | Dropped offers | +| --- | ---: | ---: | ---: | ---: | ---: | +| Cellule write, before | 512.02 | 853.65 | 751.90 | 30,367 | 58,688 | +| Cellule write, candidate | 550.52 | 866.06 | 433.80 | 52,155 | 34,687 | +| celld write | 1,994.57 | 14.29 | 14.30 | 0 | 318 | +| Cellule read-only, before | 13,780.35 | 48.28 | 48.30 | 0 | 373,130 | +| Cellule read-only, candidate | 17,877.42 | 31.04 | 31.10 | 0 | 127,309 | +| celld read-only | 19,522.95 | 8.45 | 8.50 | 0 | 28,589 | + +Write throughput is 7.52% higher in this pair; successful scheduled p99 is +1.45% higher. Read throughput is 29.73% higher. The matched read ratios are +1.297 for throughput and 0.644 for all-attempt scheduled p99, but the canonical +read guardrail still fails because both arms fail delivery. No repeatable or +attributable gain is established by one short pair. The preceding result from +the same before binary was 551.73 writes/s and 16,631.43 reads/s; it remains a +separate historical observation. These offered points do not establish either +system's maximum capacity. + +Write warmup errors/drops are 18,813/30,388 before, 10,983/24,978 candidate and +3/0 celld. Read warmup errors are zero; drops are 453,468/426,220/34,379. +Warmup failures remain qualification failures. All Cellule returned write errors +preserve HTTP 503 (`Cell is temporarily unavailable`). Celld's three warmup +errors preserve HTTP 400 (`request identity expired or not yet valid`). Every +write error class and count reconciles with all 128 client journals per case. Typed per-Cell pressure causes +remain unresolved; the HTTP failure alone does not identify that transition. + +| Case | Complete ACK cohort | Warm reads / original retries | Joined drain | Cold reads / original retries | +| --- | ---: | --- | ---: | --- | +| Cellule write, before | 43,745 | All pass | 40.45 s | All pass | +| Cellule write, candidate | 59,198 | All pass | 45.83 s | All pass | +| celld write | 181,680 | All pass | 21.30 s | All pass | +| Cellule read-only, before | 2,001 | All pass | 45.78 s | All pass | +| Cellule read-only, candidate | 2,001 | All pass | 40.76 s | All pass | +| celld read-only | 2,001 | All pass | 4.88 s | All pass | + +Independent replay reconciles every warm/timed attempt, completion, error, +dropped offer, per-Cell distribution, ACK count, receipt identity and expected +read output. All ACK mutations and original retries pass both audits with zero +errors or changed incarnations. Earlier failed initialization, warm-availability +and recovery attempts remain evidence; these passing cases do not erase them. + +## Why the shared components still have different performance + +[`assign_capture`](../crates/cellule-runtime/src/node/log_shipper/mod.rs) +reserves publication capacity while holding the global ordered issuance lock, +before committing a ticket and enqueueing follower work. A slow publication +consumer therefore queues otherwise independent Cells on the Fleet path. + +```mermaid +flowchart LR + A[Cellule SQL commit] --> B[Wait for publication credit under issuance lock] + B --> C[Assign and enqueue follower batch] + C --> D[Recoverable follower proof and Fleet ACK] + E[Verify and select shared bundle] --> F[Release capture credit] + F --> B +``` + +| Successful submission phase, mean ms | Before | Candidate | +| --- | ---: | ---: | +| Local load | 0.09629 | 0.11347 | +| Global ordered-lock wait | 142.99647 | 107.11040 | +| Publication capacity with lock held | 1.51152 | 1.60629 | +| Ticket assignment and enqueue | 0.00661 | 0.00731 | +| Complete submission | 144.61130 | 108.83792 | + +All eight phase counts reconcile with 30,823 before and 33,143 candidate +successful assignments; the partition residual is zero. The candidate still +spends 98.41% of submission time waiting for ordering. SQL-worker and +follower-proof means are 0.206 and 33.659 ms in separate overlapping cohorts; +these are not additive to that partition or measures of CPU utilization. + +Window GET/range attempts per completed write are 6.35 before and 7.67 +candidate; successful PUTs are 0.289 and 0.379. Materialized commands/root +are 8.43 and 14.86. These include background work and exclude SDK retries. +This change reduces encoder scanning, not object-store request count. No total +publication amplification reduction is demonstrated. + +Candidate pending publications remain 2,392 at both boundaries; retained bytes +grow 64,502,946→65,222,836 against 67,108,864. Unpublished node-log bytes grow +28,671,326→45,465,827, while oldest unpublished age is 48.11→48.25 seconds. +The node remains unfenced with active Fleet and all 2,000 Cells. Two endpoints +cannot qualify a sustained debt slope. Read-only windows also materialize +1,085 before and 1,345 candidate seed roots, so they include background debt. + +The preceding fine timing probe at `7d52e88` measured about 52 captures per +93.85-ms selection, including 18.48-ms catalog loading, 11.63-ms encoding, +23.80-ms PUT, 15.61-ms base verification and 10.64-ms historical verification. +Those nested phases are not all additive. About 52/0.094 ≈ 550 captures/s +before checkpoints explains the observed order of magnitude. That instrumented +probe also failed warm availability; it is diagnosis, not this candidate's +production throughput or complete recovery evidence. + +Celld's [Fleet loop](https://github.com/denoland/celld/blob/f2bf648663a610eefde71f3547ad61e9b896b1f0/crates/celld/ltx_repl.rs#L4514) +pipelines rounds in order independently of bucket publication. Its +[follower stream](https://github.com/denoland/celld/blob/f2bf648663a610eefde71f3547ad61e9b896b1f0/crates/celld/node_log.rs#L160) +groups delivered frames before durable append. Cellule currently awaits each +append batch. SQLite WAL, LTX, followers and object storage are shared +components; admission, scheduling and publication work remain different. + +## Remaining work and verification + +Prioritize reduced catalog/base/history and upload work, recoverably bounded +separation of native admission from publication debt, and ordered follower +pipelining. Larger queues or weaker proof checks do not supply sustainable +capacity. Diagnose overload availability and read variance; complete Bucket +adapter integration, mixed load, failed-owner lifecycle and safe collection. +Keep the [capacity contract](../crates/cellule-runtime/docs/write-performance-design.md): +three paired five-minute repetitions, zero errors/drops, Fleet p99 ≤50 ms, +Bucket p99 ≤200 ms, bounded debt and all-ACK cold recovery on the specified node. + +All contributor routes pass in the exact frozen production source: 1,981 +workspace tests, 60 local LTX tests, 95 bundle tests, the new regression in three +repeats, Clippy with warnings denied on Rust 1.97 and 1.99, all-feature checks, +format, API docs and boundary/layout/document/SQL-peer gates. The 38 ignored +tests retain their documented environments. Differential source and debug +measurements remain outside Git; they are not application qualification. + +Raw snapshots, exact build identities, failures, binaries, journals, metrics, +audits and inventories remain outside Git under +`/Volumes/Workspace/crabbuild-target/cellule-write-perf-8ad1/encoder-cost-20261009-*`. +All six canonical reports and their comparison explicitly fail qualification. + +[Previous comparison](pr67-catalog-overlap-measurement.md), +[implementation](bundle-coverage-implementation.md), +[delivery](write-performance-delivery.md). diff --git a/docs/write-performance-delivery.md b/docs/write-performance-delivery.md index 47b4cb76..96a53573 100644 --- a/docs/write-performance-delivery.md +++ b/docs/write-performance-delivery.md @@ -6,14 +6,17 @@ comparison retains exact retry and cold-state audits. **Celld write parity has not been established.** The [proposal](write-performance-proposal.md) remains the acceptance contract; completing tests or a load run does not pass its gates. -The latest [paired 2,000-Cell comparison](pr67-catalog-overlap-measurement.md) -measures `98a368a` at **551.73 Fleet writes/s versus 211.37 before and 1,999.88 -for fresh celld**. Read-only throughput falls 18,778.07→16,631.43/s versus -celld's 19,994.58/s. All six cases pass joined drain and complete ACK warm/cold -read/retry audits. The candidate still returns 38,642 write errors, drops offers -and accumulates publication debt; the matched read guardrail fails. One short -pair establishes no repeatable improvement or regression. Performance parity -remains unmet and PR #67 stays a draft. Earlier failures below remain evidence. +The latest [paired 2,000-Cell comparison](pr67-encoder-cost-measurement.md) +measures `30cd960` at **550.52 Fleet writes/s versus 512.02 before and 1,994.57 +for fresh celld**. Successful scheduled write p99 rises 853.65→866.06 ms; +returned errors increase and drops decrease. Read-only throughput is +17,877.42/s versus 13,780.35 before and celld's 19,522.95/s. All six cases pass +joined drain and complete ACK warm/cold read/retry audits. Publication cost +and debt remain, both read arms drop offers, and every performance profile +fails. This single short pair establishes no repeatable or attributable gain. +PR #67 remains a draft. The +[preceding catalog-overlap comparison](pr67-catalog-overlap-measurement.md) +and earlier failures below remain evidence. The earlier [selection-readiness comparison](pr67-selection-readiness-measurement.md) measures `6c909a6` at **184.77 Fleet writes/s and 248.68 Bucket writes/s**, versus From 4a5b00148ba579c1f4e035b16e8eb783c3d6fe6d Mon Sep 17 00:00:00 2001 From: forhappy Date: Fri, 9 Oct 2026 07:05:54 -0700 Subject: [PATCH 073/102] Coalesce authenticated catalog and history range reads --- .../docs/write-performance-design.md | 29 ++- .../src/node/bundle/index/io/metadata/mod.rs | 137 +++++++++++++ .../node/bundle/index/io/metadata/tests.rs | 90 ++++++++ .../node/bundle/index/{io.rs => io/mod.rs} | 194 ++++++++++-------- .../bundle/tests/index/metadata_cohort.rs | 42 ++-- .../bundle/tests/index/metadata_windows.rs | 115 +++++++++++ .../src/node/bundle/tests/index/mod.rs | 1 + 7 files changed, 496 insertions(+), 112 deletions(-) create mode 100644 crates/cellule-runtime/src/node/bundle/index/io/metadata/mod.rs create mode 100644 crates/cellule-runtime/src/node/bundle/index/io/metadata/tests.rs rename crates/cellule-runtime/src/node/bundle/index/{io.rs => io/mod.rs} (71%) create mode 100644 crates/cellule-runtime/src/node/bundle/tests/index/metadata_windows.rs diff --git a/crates/cellule-runtime/docs/write-performance-design.md b/crates/cellule-runtime/docs/write-performance-design.md index 2e6d660b..08019f19 100644 --- a/crates/cellule-runtime/docs/write-performance-design.md +++ b/crates/cellule-runtime/docs/write-performance-design.md @@ -99,14 +99,27 @@ bodies beyond the operation. Working admission remains 20 MiB; workload retention and protocol bounds remain unchanged. This is a structural charge, not allocator-profile qualification. -Catalog shard reads now overlap at most eight raw bodies after the complete -selected-shard byte preflight. Requested detached histories use a separate -eight-read phase after the shard jobs join and their aggregate byte preflight -passes. Both phases retain the original 4-MiB selected-metadata bound and serial -authenticated decoding. Planning keeps at most 256 one-byte shard IDs and eight -history indices; unrelated histories remain authenticated references. This -changes scheduling rather than request count or proof policy. Its -[fresh paired diagnostic](../../../docs/pr67-catalog-overlap-measurement.md) +Catalog loading coalesces contiguous same-object shard extents after the +complete selected-shard byte preflight. Requested detached histories use a +separate phase after the shard jobs join and the combined metadata preflight +passes. History windows may bridge gaps of at most 32 KiB, charging every gap +once against the unused original 4-MiB selected-metadata allowance. Shard +windows spend no gap credit. Exhausted credit preserves separate valid reads. +Each phase joins at most eight windows before canonical decoding of every +original authenticated extent. Padding confers no authority; unrelated +histories remain references. Planning retains at most 256 shard and 4,096 +history indices as compact `u16` values, plus eight window descriptors within +the existing structural working charge. No availability cache, producer +reservation increase or protocol change is introduced. Coalescing can trade +more transferred bytes for fewer requests; measure both costs. + +The real 64-Cell, 2,000-binding preparation regression needs 123 metadata +reads on the unchanged loader and three after coalescing. Three repetitions +reproduce each result; every participating Cell cold-restores its exact seed +and outcomes. Separate-object fixtures continue to enforce eight reads with +no ninth read, cancellation without publication and missing-origin rejection. +These are component results, not new application TPS or qualification. The +preceding [catalog overlap diagnostic](../../../docs/pr67-catalog-overlap-measurement.md) records a higher write rate and lower read rate; performance qualification fails. Dense encoding now maps the original at-most-64 frame digests to unique diff --git a/crates/cellule-runtime/src/node/bundle/index/io/metadata/mod.rs b/crates/cellule-runtime/src/node/bundle/index/io/metadata/mod.rs new file mode 100644 index 00000000..8b250b09 --- /dev/null +++ b/crates/cellule-runtime/src/node/bundle/index/io/metadata/mod.rs @@ -0,0 +1,137 @@ +//! Bounded windows retain compact indices and authenticate every original extent. +use super::*; +use std::ops::Range; + +pub(super) struct Window { + pub(super) indices: Range, + object: Digest, + range: Range, +} + +pub(super) fn sort<'a>( + indices: &mut [u16], + extent: impl Fn(u16) -> Result<&'a Locator>, +) -> Result<()> { + for index in indices.iter().copied() { + let extent = extent(index)?; + codec::validate_extent(extent)?; + extent + .object + .ok_or(Error::Node("unresolved bundle metadata"))?; + } + let mut failed = None; + indices.sort_unstable_by(|left, right| { + let keys = extent(*left).and_then(|left| { + extent(*right).map(|right| { + (left.object.map(|object| *object.as_bytes()), left.offset) + .cmp(&(right.object.map(|object| *object.as_bytes()), right.offset)) + }) + }); + match keys { + Ok(order) => order, + Err(error) => { + failed = Some(error); + std::cmp::Ordering::Equal + } + } + }); + if let Some(error) = failed { + return Err(error); + } + Ok(()) +} + +pub(super) fn cohort<'a>( + indices: &[u16], + next: &mut usize, + padding: &mut u64, + extent: impl Fn(u16) -> Result<&'a Locator>, +) -> Result> { + let mut windows = Vec::with_capacity(READ_CONCURRENCY); + while windows.len() < READ_CONCURRENCY && *next < indices.len() { + let first = *next; + let initial = extent(indices[first])?; + let object = initial + .object + .ok_or(Error::Node("unresolved bundle metadata"))?; + let mut range = initial.offset + ..initial + .offset + .checked_add(initial.bytes) + .ok_or(Error::Node("bundle metadata extent overflow"))?; + *next += 1; + while let Some(index) = indices.get(*next) { + let candidate = extent(*index)?; + let gap = candidate.offset.saturating_sub(range.end); + if candidate.object != Some(object) + || gap > *padding + || gap > history::MAX_HISTORY_BYTES + { + break; + } + let end = candidate + .offset + .checked_add(candidate.bytes) + .ok_or(Error::Node("bundle metadata extent overflow"))?; + // A gap consumes byte credit once. Overlapping extents cannot buy + // more padding, and every requested extent remains in the window. + *padding -= gap; + range.end = range.end.max(end); + *next += 1; + } + windows.push(Window { + indices: first..*next, + object, + range, + }); + } + Ok(windows) +} + +pub(super) async fn read( + layout: &cellule_ltx::CellStorageLayout, + root: &Root, + origin: Option<&super::super::super::origin::OriginBundle>, + window: Window, +) -> Result<(Window, Bytes)> { + let bytes = super::super::super::origin::read_range( + layout, + root.session, + root.epoch, + window.object, + window.range.clone(), + origin, + ) + .await?; + if bytes.len() as u64 != window.range.end - window.range.start { + return Err(Error::Node("bundle metadata window is truncated")); + } + Ok((window, bytes)) +} + +impl Window { + pub(super) fn slice(&self, body: &Bytes, extent: &Locator) -> Result { + let start = extent + .offset + .checked_sub(self.range.start) + .and_then(|start| usize::try_from(start).ok()); + let end = start.and_then(|start| { + usize::try_from(extent.bytes) + .ok() + .and_then(|bytes| start.checked_add(bytes)) + }); + let (start, end) = start + .zip(end) + .filter(|(_, end)| { + *end <= body.len() && *end as u64 <= self.range.end - self.range.start + }) + .ok_or(Error::Node("bundle metadata extent is truncated"))?; + if extent.object != Some(self.object) { + return Err(Error::Node("bundle metadata window scope differs")); + } + Ok(body.slice(start..end)) + } +} + +#[cfg(test)] +mod tests; diff --git a/crates/cellule-runtime/src/node/bundle/index/io/metadata/tests.rs b/crates/cellule-runtime/src/node/bundle/index/io/metadata/tests.rs new file mode 100644 index 00000000..ad118064 --- /dev/null +++ b/crates/cellule-runtime/src/node/bundle/index/io/metadata/tests.rs @@ -0,0 +1,90 @@ +use super::*; + +fn extent(object: u8, start: u64, bytes: u64) -> Locator { + Locator { + object: Some(Digest::from_bytes([object; 32])), + offset: HEADER_BYTES as u64 + start, + bytes, + frame_digest: Digest::from_bytes([9; 32]), + } +} + +#[test] +fn sparse_metadata_windows_charge_gaps_once_and_preserve_all_original_indices() { + let extents = [ + extent(1, 0, 10), + extent(1, 5, 10), + extent(1, 25, 10), + extent(1, 50, 10), + extent(2, 0, 10), + ]; + let locate = |index: u16| Ok(&extents[usize::from(index)]); + let mut indices = [4, 3, 2, 1, 0]; + sort(&mut indices, locate).unwrap(); + let mut next = 0; + let mut padding = 10; + let windows = cohort(&indices, &mut next, &mut padding, locate).unwrap(); + assert_eq!(windows.len(), 3); + assert_eq!(windows[0].indices, 0..3); + assert_eq!( + windows[0].range, + HEADER_BYTES as u64..HEADER_BYTES as u64 + 35 + ); + assert_eq!(windows[1].indices, 3..4); + assert_eq!(windows[2].object, Digest::from_bytes([2; 32])); + assert_eq!(padding, 0, "overlap cannot buy gap credit"); + assert_eq!(next, extents.len()); + let wire: u64 = windows.iter().map(|w| w.range.end - w.range.start).sum(); + assert!(wire <= extents.iter().map(|e| e.bytes).sum::() + 10); + let body = Bytes::from_static(b"01234567890123456789012345678901234"); + assert_eq!( + windows[0].slice(&body, &extents[2]).unwrap(), + body.slice(25..35) + ); + assert!(windows[0].slice(&body.slice(..10), &extents[2]).is_err()); + assert!(windows[0].slice(&body, &extents[4]).is_err()); +} + +#[test] +fn eight_metadata_windows_and_compact_protocol_indices_stay_bounded() { + let extents: Vec<_> = (1..=12).map(|object| extent(object, 0, 10)).collect(); + let locate = |index: u16| Ok(&extents[usize::from(index)]); + let mut indices: Vec<_> = (0..extents.len()).map(|index| index as u16).collect(); + sort(&mut indices, locate).unwrap(); + let mut next = 0; + let mut padding = MAX_BUNDLE_BYTES; + assert_eq!( + cohort(&indices, &mut next, &mut padding, locate) + .unwrap() + .len(), + 8 + ); + assert_eq!(next, 8); + assert_eq!( + cohort(&indices, &mut next, &mut padding, locate) + .unwrap() + .len(), + 4 + ); + assert_eq!(padding, MAX_BUNDLE_BYTES); + // Catalog and history phases do not overlap. Bound each phase's retained + // index vector and eight descriptors, leaving 16 KiB for task overhead. + let planning = MAX_BINDINGS * std::mem::size_of::() + + READ_CONCURRENCY * std::mem::size_of::(); + assert!(planning + 16 * 1024 <= 32 * 1024); +} + +#[test] +fn exhausted_gap_credit_keeps_valid_metadata_as_separate_reads() { + let extents = [extent(1, 0, 10), extent(1, 11, 10)]; + let locate = |index: u16| Ok(&extents[usize::from(index)]); + let mut indices = [0, 1]; + sort(&mut indices, locate).unwrap(); + let mut next = 0; + let mut padding = 0; + let windows = cohort(&indices, &mut next, &mut padding, locate).unwrap(); + assert_eq!(windows.len(), 2); + let mut invalid = [extent(1, 0, 1)]; + invalid[0].offset = u64::MAX; + assert!(sort(&mut [0], |index| Ok(&invalid[usize::from(index)])).is_err()); +} diff --git a/crates/cellule-runtime/src/node/bundle/index/io.rs b/crates/cellule-runtime/src/node/bundle/index/io/mod.rs similarity index 71% rename from crates/cellule-runtime/src/node/bundle/index/io.rs rename to crates/cellule-runtime/src/node/bundle/index/io/mod.rs index 2c98bf68..14b6578f 100644 --- a/crates/cellule-runtime/src/node/bundle/index/io.rs +++ b/crates/cellule-runtime/src/node/bundle/index/io/mod.rs @@ -1,6 +1,8 @@ //! Bounded origin point lookups and streaming maintenance inventory checks. use super::*; -use futures_util::{StreamExt, future::try_join_all, stream}; +use futures_util::future::try_join_all; + +mod metadata; const READ_CONCURRENCY: usize = 8; type DecodedRows = (Vec, BTreeMap<[u8; 32], history::History>); @@ -160,42 +162,51 @@ async fn load_inner( loaded.insert(id, Vec::new()); } } - // Aggregate encoded bytes were checked before opening any read. At most - // eight raw bodies coexist within that same bound; decode stays serial. - // A fixed-header-sized list retains at most 256 one-byte shard IDs and no - // decoded siblings. Owned indices keep this future Send at authority seams. + // Selected shard bytes were preflighted before any read. Shard windows + // spend no gap credit: requested history sizes are still unknown here. + // Up to eight joined windows retain the same raw-body byte allowance. let mut shards = Vec::with_capacity(SHARDS); for (id, shard) in root.shards.iter().enumerate() { if shard.is_some() && !wanted.is_some_and(|wanted| !wanted.contains(&(id as u8))) { - shards.push(id as u8); + shards.push(id as u16); } } - let mut reads = stream::iter(shards) - .map(|id| { - let root = &root; - async move { + let shard_extent = |id: u16| { + root.shards + .get(usize::from(id)) + .and_then(Option::as_ref) + .map(|shard| &shard.extent) + .ok_or(Error::Node("bundle shard plan is absent")) + }; + metadata::sort(&mut shards, shard_extent)?; + let mut next = 0; + let mut padding = 0; + while next < shards.len() { + let windows = metadata::cohort(&shards, &mut next, &mut padding, shard_extent)?; + let bodies = try_join_all( + windows + .into_iter() + .map(|window| metadata::read(layout, &root, origin, window)), + ) + .await?; + for (window, bytes) in bodies { + for index in window.indices.clone() { + let id = shards[index] as u8; let shard = root.shards[usize::from(id)] .as_ref() .ok_or(Error::Node("bundle shard plan is absent"))?; - Ok::<_, Error>((id, read_rows(layout, root, shard, origin).await?)) - } - }) - .buffered(READ_CONCURRENCY); - while let Some(result) = reads.next().await { - let (id, bytes) = result?; - let shard = root.shards[usize::from(id)] - .as_ref() - .ok_or(Error::Node("bundle shard plan is absent"))?; - let (rows, leaf_histories) = decode_rows(&root, shard, id, bytes)?; - for (pin, history) in leaf_histories { - if histories.insert(pin, history).is_some() { - return Err(Error::Node("bundle inventory repeats a Cell pin")); + let body = window.slice(&bytes, &shard.extent)?; + let (rows, leaf_histories) = decode_rows(&root, shard, id, body)?; + for (pin, history) in leaf_histories { + if histories.insert(pin, history).is_some() { + return Err(Error::Node("bundle inventory repeats a Cell pin")); + } + } + bindings.extend(rows.iter().cloned()); + loaded.insert(id, rows); } } - bindings.extend(rows.iter().cloned()); - loaded.insert(id, rows); } - drop(reads); bindings.sort_unstable_by_key(|binding| { binding .control @@ -221,7 +232,16 @@ async fn load_inner( .ok_or(Error::Capacity("bundle selected history bytes"))?; } } - hydrate_histories(layout, &root, cells, origin, &mut bindings, &mut histories).await?; + hydrate_histories( + layout, + &root, + cells, + origin, + MAX_BUNDLE_BYTES - selected_bytes, + &mut bindings, + &mut histories, + ) + .await?; // Snapshot after requested histories have loaded: hydration itself must // not rewrite a shard that the caller never changes. let mut hydrated = BTreeMap::>::new(); @@ -250,83 +270,91 @@ async fn load_inner( Ok(catalog) } +#[allow( + clippy::too_many_arguments, + reason = "the original aggregate preflight supplies this phase's gap credit" +)] async fn hydrate_histories( layout: &cellule_ltx::CellStorageLayout, root: &Root, cells: Option<&BTreeSet>, origin: Option<&super::super::origin::OriginBundle>, + mut padding: u64, bindings: &mut [Binding], histories: &mut BTreeMap<[u8; 32], history::History>, ) -> Result<()> { - let mut next = 0; - while next < bindings.len() { - // Only eight compact indices are planned. The already checked shard - // plus requested-history aggregate bounds every returned raw body; - // shard reads have joined before this phase reuses their allowance. - let mut targets = Vec::with_capacity(READ_CONCURRENCY); - while targets.len() < READ_CONCURRENCY && next < bindings.len() { - let binding = &bindings[next]; - if !cells.is_some_and(|cells| { - !cells.contains(&( - *binding.application.as_bytes(), - *binding.control.cell.as_bytes(), - )) - }) && histories.contains_key(&history::pin(binding)?) - { - targets.push(next); - } - next += 1; + let mut targets = Vec::with_capacity(bindings.len()); + for (index, binding) in bindings.iter().enumerate() { + if !cells.is_some_and(|cells| { + !cells.contains(&( + *binding.application.as_bytes(), + *binding.control.cell.as_bytes(), + )) + }) && histories.contains_key(&history::pin(binding)?) + { + targets.push( + u16::try_from(index) + .map_err(|_| Error::Capacity("bundle history planning indices"))?, + ); } - let borrowed_bindings = &*bindings; - let borrowed_histories = &*histories; - let bodies = try_join_all((0..targets.len()).map(|offset| { - let index = targets[offset]; - async move { - let history = borrowed_histories - .get(&history::pin(&borrowed_bindings[index])?) + } + metadata::sort(&mut targets, |index| { + history_extent(bindings, histories, index) + })?; + let mut next = 0; + while next < targets.len() { + // Only eight window descriptors and at most MAX_BINDINGS compact u16 + // indices are retained. Gaps charge the unused original shard/history + // aggregate; this phase joins after all shard bodies have dropped. + let windows = metadata::cohort(&targets, &mut next, &mut padding, |index| { + history_extent(bindings, histories, index) + })?; + let bodies = try_join_all( + windows + .into_iter() + .map(|window| metadata::read(layout, root, origin, window)), + ) + .await?; + for (window, bytes) in bodies { + for target in window.indices.clone() { + let binding = &mut bindings[usize::from(targets[target])]; + let history = histories + .get_mut(&history::pin(binding)?) .ok_or(Error::Node("bundle history plan is absent"))?; let object = history .extent .object .ok_or(Error::Node("unresolved bundle history"))?; - let body = super::super::origin::read_range( - layout, - root.session, - root.epoch, - object, - history.extent.offset..history.extent.offset + history.extent.bytes, - origin, - ) - .await?; - if body.len() as u64 != history.extent.bytes { - return Err(Error::Node("bundle history digest differs")); - } - Ok(body) - } - })) - .await?; - for (index, bytes) in targets.into_iter().zip(bodies) { - let binding = &mut bindings[index]; - let history = histories - .get_mut(&history::pin(binding)?) - .ok_or(Error::Node("bundle history plan is absent"))?; - let object = history - .extent - .object - .ok_or(Error::Node("unresolved bundle history"))?; - let mut locators = history::decode(&bytes, root.session, root.epoch, binding, history)?; - for locator in &mut locators { - if locator.object.is_none() { - locator.object = Some(object); + let body = window.slice(&bytes, &history.extent)?; + let mut locators = + history::decode(&body, root.session, root.epoch, binding, history)?; + for locator in &mut locators { + if locator.object.is_none() { + locator.object = Some(object); + } } + history.loaded = Some(locators.clone()); + binding.locators = locators; } - history.loaded = Some(locators.clone()); - binding.locators = locators; } } Ok(()) } +fn history_extent<'a>( + bindings: &[Binding], + histories: &'a BTreeMap<[u8; 32], history::History>, + index: u16, +) -> Result<&'a Locator> { + let binding = bindings + .get(usize::from(index)) + .ok_or(Error::Node("bundle history plan is absent"))?; + histories + .get(&history::pin(binding)?) + .map(|history| &history.extent) + .ok_or(Error::Node("bundle history plan is absent")) +} + /// A clean maintenance result covers every indexed binding. Only one bounded /// shard's decoded rows are retained at a time; the global duplicate-pin set is /// bounded by MAX_BINDINGS. No missing shard is interpreted as an empty shard. diff --git a/crates/cellule-runtime/src/node/bundle/tests/index/metadata_cohort.rs b/crates/cellule-runtime/src/node/bundle/tests/index/metadata_cohort.rs index f967aaf8..f2737ab3 100644 --- a/crates/cellule-runtime/src/node/bundle/tests/index/metadata_cohort.rs +++ b/crates/cellule-runtime/src/node/bundle/tests/index/metadata_cohort.rs @@ -25,26 +25,26 @@ async fn held_metadata(mode: u8) { .map(|cell| catalog_index::shard(&[9; 16], cell.control.value().cell.as_bytes())) .collect::>(); assert!(shards.len() > 8); - let mut frames = Vec::new(); - let mut assigned = Vec::new(); + // Separate immutable sources preserve the concurrency requirement after + // same-object metadata is coalesced. Every requested history has its own + // source and more than eight selected shards have distinct latest sources. + let now = f.node.advertisement().issued_at_ms(); for cell in &mut cells { - let (_, capture, range) = f.append(cell, 2); - frames.extend(capture); - assigned.push(range); + let (_, frames, assigned) = f.append(cell, 2); + let first = f + .directory + .prepare_node_bundle(&f.node, &frames, &[assigned], now) + .await + .unwrap(); + f.node = f + .directory + .select_node_bundle(&f.node, &first, &f.lease, Limits::default(), now) + .await + .unwrap() + .0; } - let now = f.node.advertisement().issued_at_ms(); - let first = f - .directory - .prepare_node_bundle(&f.node, &frames, &assigned, now) - .await - .unwrap(); - f.node = f - .directory - .select_node_bundle(&f.node, &first, &f.lease, Limits::default(), now) - .await - .unwrap() - .0; - let catalog = load_catalog(&f.layout, head_session(&f), first.head) + let head = f.node.advertisement().bundle_head().unwrap(); + let catalog = load_catalog(&f.layout, head_session(&f), head) .await .unwrap(); for cell in &cells { @@ -54,15 +54,15 @@ async fn held_metadata(mode: u8) { f.layout .node_coverage_bundle_path( head_session(&f).as_bytes(), - first.head.epoch, + head.epoch, history.object.unwrap().as_bytes(), ) .to_string(), history.offset, )); } - frames.clear(); - assigned.clear(); + let mut frames = Vec::new(); + let mut assigned = Vec::new(); for cell in &mut cells { let (_, capture, range) = f.append(cell, 3); frames.extend(capture); diff --git a/crates/cellule-runtime/src/node/bundle/tests/index/metadata_windows.rs b/crates/cellule-runtime/src/node/bundle/tests/index/metadata_windows.rs new file mode 100644 index 00000000..b2ab14f9 --- /dev/null +++ b/crates/cellule-runtime/src/node/bundle/tests/index/metadata_windows.rs @@ -0,0 +1,115 @@ +use super::*; + +#[tokio::test] +async fn preparation_reads_one_catalog_window_and_one_history_window_for_a_shared_cohort() { + let mut f = Fixture::new().await; + super::super::coverage::enroll(&mut f).await; + let mut cells = Vec::new(); + for number in 4..68 { + cells.push(f.cell(number).await); + f.heartbeat().await; + } + inventory(&mut f, &cells[0], 1_937).await; + let mut frames = Vec::new(); + let mut assigned = Vec::new(); + for cell in &mut cells { + let (_, capture, range) = f.append(cell, 2); + frames.extend(capture); + assigned.push(range); + } + let now = f.node.advertisement().issued_at_ms(); + let first = f + .directory + .prepare_node_bundle(&f.node, &frames, &assigned, now) + .await + .unwrap(); + f.node = f + .directory + .select_node_bundle(&f.node, &first, &f.lease, Limits::default(), now) + .await + .unwrap() + .0; + let path = f.layout.node_coverage_bundle_path( + head_session(&f).as_bytes(), + first.head.epoch, + first.head.digest.as_bytes(), + ); + frames.clear(); + assigned.clear(); + for cell in &mut cells { + let (_, capture, range) = f.append(cell, 3); + frames.extend(capture); + assigned.push(range); + } + f.count.reset(); + let next = f + .directory + .prepare_node_bundle(&f.node, &frames, &assigned, now) + .await + .unwrap(); + let requests = f + .count + .requests() + .into_iter() + .filter(|request| request.location == path.as_ref()) + .collect::>(); + eprintln!( + "shared metadata: cells={} reads={}", + cells.len(), + requests.len() + ); + assert_eq!( + requests.len(), + 3, + "one fresh header, contiguous catalog window and contiguous history window" + ); + let (_, proofs) = f + .directory + .select_node_bundle(&f.node, &next, &f.lease, Limits::default(), now) + .await + .unwrap(); + assert_eq!(proofs.len(), cells.len()); + for proof in &proofs { + assert_eq!(proof.commit_sequence(), 3); + let cell = cells + .iter() + .find(|cell| cell.control.value().bundle_binding == Some(proof.binding())) + .unwrap(); + let overlay = proof + .recovery_overlay(&f.layout, Limits::default()) + .await + .unwrap(); + let recovered = cell + .replica + .prepare_recovered_overlay(&overlay, 1) + .await + .unwrap(); + let cold = f.scratch.path().join(format!( + "metadata-window-{}.sqlite", + cell.control.value().cell.as_bytes()[0] + )); + cell.replica + .open_root(&recovered.root()) + .await + .unwrap() + .restore(&cold) + .await + .unwrap(); + let db = rusqlite::Connection::open(cold).unwrap(); + let outcomes: Vec<(String, String)> = db + .prepare("SELECT request, result FROM outcomes ORDER BY request") + .unwrap() + .query_map([], |row| Ok((row.get(0)?, row.get(1)?))) + .unwrap() + .collect::>() + .unwrap(); + assert_eq!( + outcomes, + vec![ + ("request-2".into(), "result-2".into()), + ("request-3".into(), "result-3".into()), + ("seed".into(), "original".into()) + ] + ); + } +} diff --git a/crates/cellule-runtime/src/node/bundle/tests/index/mod.rs b/crates/cellule-runtime/src/node/bundle/tests/index/mod.rs index 55ceb6fa..158abc16 100644 --- a/crates/cellule-runtime/src/node/bundle/tests/index/mod.rs +++ b/crates/cellule-runtime/src/node/bundle/tests/index/mod.rs @@ -49,3 +49,4 @@ mod encoding; mod history_cohort; mod inventory; mod metadata_cohort; +mod metadata_windows; From c64d3ad787cd11f922eb2e4d0e3fa580b552b1ae Mon Sep 17 00:00:00 2001 From: forhappy Date: Fri, 9 Oct 2026 07:45:31 -0700 Subject: [PATCH 074/102] Record metadata window costs and failed parity measurements --- .../docs/write-performance-design.md | 29 +-- docs/bundle-coverage-implementation.md | 27 +-- docs/pr67-metadata-window-measurement.md | 187 ++++++++++++++++++ docs/write-performance-delivery.md | 23 +-- 4 files changed, 228 insertions(+), 38 deletions(-) create mode 100644 docs/pr67-metadata-window-measurement.md diff --git a/crates/cellule-runtime/docs/write-performance-design.md b/crates/cellule-runtime/docs/write-performance-design.md index 08019f19..ee19fbd1 100644 --- a/crates/cellule-runtime/docs/write-performance-design.md +++ b/crates/cellule-runtime/docs/write-performance-design.md @@ -1,21 +1,22 @@ # Node write and read performance design -The [current paired 2,000-Cell measurements](../../../docs/pr67-encoder-cost-measurement.md) -complete joined drain and all-ACK warm/cold read and original-retry audits in -all six cases. Reduced encoder scanning completes 550.52 Fleet writes/s versus -512.02 before and 1,994.57 for fresh celld. Successful scheduled write p99 is -866.06 ms versus 853.65 before; errors increase while dropped offers decrease. -Read-only throughput is 17,877.42/s versus 13,780.35 before and 19,522.95 for -celld. Both Cellule read arms drop offers, so the canonical read guardrail fails. -One short pair establishes no repeatable or attributable gain. Publication -cost and debt remain; all profiles are unqualified and PR #67 stays a draft. -The [preceding comparison](../../../docs/pr67-catalog-overlap-measurement.md) -and its failures remain evidence. +The [current paired 2,000-Cell measurements](../../../docs/pr67-metadata-window-measurement.md) +complete 502.28 Fleet writes/s versus 455.73 before and 1,999.83 for fresh +celld. Metadata windows reduce observed GET/range work, but successful scheduled +write p99 worsens 827.14→1,454.42 ms and returned errors increase. Read-only +throughput falls 17,905.02→17,347.03/s; celld completes 19,983.30/s. The candidate +passes all 48,888 ACK warm/cold reads and original retries and drains in 66.52 +seconds. The baseline write case fails two warm retries and never reaches the +cold audit. Both Cellule read arms drop offers and the read guardrail fails. +One short pair establishes no acceptable, repeatable or attributable gain. +Publication cost and debt remain; all profiles are unqualified and PR #67 stays +a draft. The [preceding encoder comparison](../../../docs/pr67-encoder-cost-measurement.md) +and its passing and failed evidence remain separate observations. The [submission diagnosis](../../../docs/pr67-submission-timing-measurement.md) -identified publication capacity held under the global issuance lock. The latest -candidate still spends 98.41% of successful submission time waiting for that -lock. This coupling queues native progress before follower proof starts. Repeated +identified publication capacity held under the global issuance lock. Observed ordered wait remains about 119 ms; the latest phase counts +straddle window boundaries and cannot support an exact successful-submission +partition. This coupling queues native progress before follower proof starts. Repeated historical/base verification and sparse root checkpoints keep the publication consumer expensive. Matching celld requires reducing that work and separating native progress from bounded recoverable publication debt; larger queues alone diff --git a/docs/bundle-coverage-implementation.md b/docs/bundle-coverage-implementation.md index 72088a58..7fe248e5 100644 --- a/docs/bundle-coverage-implementation.md +++ b/docs/bundle-coverage-implementation.md @@ -1,20 +1,21 @@ # Bundle coverage implementation -The [current paired 2,000-Cell measurements](pr67-encoder-cost-measurement.md) -complete joined drain and all-ACK warm/cold read and original-retry audits in -all six cases. Reduced encoder scanning completes 550.52 Fleet writes/s versus -512.02 before and 1,994.57 for fresh celld. Successful scheduled write p99 is -866.06 ms versus 853.65 before; errors increase while dropped offers decrease. -Read-only throughput is 17,877.42/s versus 13,780.35 before and 19,522.95 for -celld. Both Cellule read arms drop offers, so the canonical read guardrail fails. -One short pair establishes no repeatable or attributable gain. Publication -cost and debt remain; all profiles are unqualified and PR #67 stays a draft. -The [preceding comparison](pr67-catalog-overlap-measurement.md) -and its failures remain evidence. +The [current paired 2,000-Cell measurements](pr67-metadata-window-measurement.md) +complete 502.28 Fleet writes/s versus 455.73 before and 1,999.83 for fresh +celld. Metadata windows reduce observed GET/range work, but successful scheduled +write p99 worsens 827.14→1,454.42 ms and returned errors increase. Read-only +throughput falls 17,905.02→17,347.03/s; celld completes 19,983.30/s. The candidate +passes all 48,888 ACK warm/cold reads and original retries and drains in 66.52 +seconds. The baseline write case fails two warm retries and never reaches the +cold audit. Both Cellule read arms drop offers and the read guardrail fails. +One short pair establishes no acceptable, repeatable or attributable gain. +Publication cost and debt remain; all profiles are unqualified and PR #67 stays +a draft. The [preceding encoder comparison](pr67-encoder-cost-measurement.md) +and its passing and failed evidence remain separate observations. The [submission diagnosis](pr67-submission-timing-measurement.md) identifies -publication capacity held under the global issuance lock. The latest candidate -still spends 98.41% of successful submission time waiting for that lock. +publication capacity held under the global issuance lock. Observed ordered wait remains about 119 ms; unequal phase counts prevent an +exact successful-submission partition in this latest window. Celld pipelines native progress independently of bucket publication. Similar components do not imply the same response path. PR #67 remains a draft. diff --git a/docs/pr67-metadata-window-measurement.md b/docs/pr67-metadata-window-measurement.md new file mode 100644 index 00000000..e9ea3895 --- /dev/null +++ b/docs/pr67-metadata-window-measurement.md @@ -0,0 +1,187 @@ +# PR 67: metadata windows and fresh paired measurements + +**Performance parity remains unmet.** The candidate completes 502.28 Fleet +writes/s versus 455.73 before and 1,999.83 for celld at the same 2,000/s offered +load. Successful scheduled write p99 worsens 827.14→1,454.42 ms, and returned +write errors increase. Read-only throughput falls 3.12% in this pair. The +candidate passes complete ACK warm/cold read and original-retry audits and +drains in 66.52 seconds. The baseline write case fails two warm retries and +does not reach the cold audit. Every performance profile fails; PR #67 remains +a draft. + +## Change and bounded verification + +The preceding loader issued separate range requests for selected catalog +shards and detached histories even when they shared an immutable object. +Loading now sorts compact indices by object and offset and reads bounded +windows. Contiguous shard extents merge without fetching gaps. History windows +may bridge gaps of at most 32 KiB, charging each gap once against the unused +original 4-MiB combined metadata allowance. Exhausted credit preserves separate +valid reads. Up to eight windows join before every original extent is sliced +and canonically authenticated. Padding supplies no coverage rights. + +Shard and history phases remain separate; shard bodies drop before history +reads. Plans retain at most 256 shard indices or 4,096 history indices as `u16` +values and eight window descriptors. No availability cache or new admission +policy is introduced. Persisted formats, authority pins, native verification, +the 20-MiB working reservation and maintenance inventory semantics remain +unchanged. Coalescing trades request count against returned bytes; both costs +are measured below. + +The real 64-Cell preparation regression has a 2,000-binding catalog, reads one +header, one catalog window and one history window, and cold-restores all 64 +Cells' exact seeds and outcomes. The unchanged loader fails its three-read +assertion in three repetitions with **123 reads**; the candidate passes all +three with **three reads**. Other catalog entries are metadata fixtures, not +2,000 active writers. Distinct-object fixtures preserve the original eight-read +limit, cancellation without publication, exact recovery and missing-origin +rejection. Those fixtures pass three repetitions before and after the change. +Three planner tests and all 99 bundle tests pass. These are component results, +not application TPS. + +## Matched workload and provenance + +| Dimension | Diagnostic | +| --- | --- | +| Before executable | `30cd960ff78a2be31fe84493ad0038b2ad4be7ad` | +| Candidate executable | `4a5b00148ba579c1f4e035b16e8eb783c3d6fe6d` | +| Celld | `f2bf648663a610eefde71f3547ad61e9b896b1f0` | +| Active population | 2,000 real Cells, uniformly offered; every Cell receives timed successful responses in every case | +| Command | 96-byte SQL INSERT/SELECT values and the same two-hour request/result ledger | +| Fleet | One owner, two followers; local state and follower logs on tmpfs | +| Arrival | 128 clients/queue slots, 30-second warmup, one 60-second window | +| Offered load | Separate 2,000-write/s and 20,000-read/s cases | +| Admission | Original 64-MiB retained budget, 1-GiB managed disk and 20-MiB producer reservation | + +All six cases are fresh, with new namespaces and provider volumes. The before +executable is a byte-identical copy of the preceding verified executable, +measured again here. The candidate is a pinned Linux release build with no +measurement overlay. Its SQL SHA-256 is +`bc1a577f3a1851432497cc3f4a6569272a225d252e9dab72195e66674b93e43c`. +All 1,940 exported framework files, including symlink targets, match the commit; +all 1,318 Rust/TOML/lock inputs match the contributor-check snapshot. Clients, +auditor, runner, workload fixtures, images and resource settings match. +Both managed SQLite paths use WAL NORMAL. No build, broad suite or journal +audit overlaps a timed window. + +The shared ARM64 Docker VM has eight CPUs and about 8 GiB total memory across +owner, followers, provider and client. Container ceilings exceed that capacity, +internal admission policies differ, and another host VM remains active. VM +settings are unchanged within the comparison. These diagnostics do not qualify +a dedicated 8-vCPU/16-GiB serving process or physical-media durability. They do +not establish maximum throughput: the write offer is capped at 2,000/s. + +## Application results + +TPS counts successful completions inside the timed window. Successful scheduled +p99 includes trailing successes; all-attempt p99 includes fast failures and is +reported separately. Dropped offers are separate from returned errors. + +| Point | Successes/s | Successful scheduled p99 ms | All-attempt scheduled p99 ms | Errors | Dropped offers | +| --- | ---: | ---: | ---: | ---: | ---: | +| Cellule write, before | 455.73 | 827.14 | 627.0 | 36,569 | 56,070 | +| Cellule write, candidate | 502.28 | 1,454.42 | 729.5 | 50,573 | 39,162 | +| Celld write | 1,999.83 | 13.60 | 13.7 | 0 | 0 | +| Cellule read-only, before | 17,905.02 | 26.87 | 26.9 | 0 | 125,645 | +| Cellule read-only, candidate | 17,347.03 | 31.26 | 31.3 | 0 | 159,122 | +| Celld read-only | 19,983.30 | 3.51 | 3.6 | 0 | 973 | + +Write throughput is 10.21% higher in this pair, but successful write p99 is +75.84% higher. Read throughput is 3.12% lower and successful read p99 is higher. +The canonical read guardrail fails: throughput ratio 0.969, all-attempt p99 +ratio 1.164, and both Cellule arms fail delivery. No acceptable, repeatable or +attributable performance gain is established. The preceding measurement of the +same baseline executable was 550.52 writes/s and 17,877.42 reads/s; that separate +observation remains evidence of run variance rather than a substitute control. + +Write warmup errors/drops are 0/32,928 before, 19,799/23,579 candidate and 0/0 +celld. Read warmup errors are zero; drops are 420,544/410,166/2,975. Every +Cellule write error preserves HTTP 503 and `Cell is temporarily unavailable`. +All 128 journals per write case reconcile exact attempt and error counts. +The response alone does not identify the underlying per-Cell pressure transition. + +| Case | ACK cohort | Warm reads / original retries | Joined drain | Cold reads / original retries | +| --- | ---: | --- | ---: | --- | +| Cellule write, before | 56,434 | Two retries fail with HTTP 503 | Not reached | Not reached | +| Cellule write, candidate | 48,888 | All pass | 66.52 s | All pass | +| Celld write | 182,001 | All pass | 16.63 s | All pass | +| Cellule read-only, before | 2,001 | All pass | 37.15 s | All pass | +| Cellule read-only, candidate | 2,001 | All pass | 40.23 s | All pass | +| Celld read-only | 2,001 | All pass | 2.80 s | All pass | + +Independent replay reconciles every timed/warm attempt, success, error, dropped +offer, per-Cell distribution, expected read output and complete ACK manifest. +Journal reconciliation is not a passing warm/cold availability audit. The +baseline's failed warm audit ends the normal drain/cold sequence; its cleanup +does not substitute for those exit gates. Every failure remains evidence. + +## Storage work and remaining coupling + +| Window cost per completed write | Before | Candidate | +| --- | ---: | ---: | +| GET/range attempts | 8.330 | 6.274 | +| Returned GET/range bytes | 64,987 | 52,515 | +| Successful PUTs | 0.395 | 0.374 | +| Materialized commands per root | 19.22 | 12.22 | + +Observed requests fall 24.68% and returned bytes fall 19.19%. These counters +include background work; SDK retries and provider wire overhead are excluded. +Starts and completions can straddle boundaries. They are not isolated command +costs or proof that coalescing caused the application TPS change. Root density +falls and publication amplification remains well above the separate cost target. + +[`assign_capture`](../crates/cellule-runtime/src/node/log_shipper/mod.rs) +still reserves publication capacity under the global ordered issuance lock +before assigning and enqueueing follower work. The publication task verifies +and selects serial cohorts before releasing capture credit. A slow publication +consumer therefore gates Fleet progress across otherwise independent Cells. +Celld's [shipping loop](https://github.com/denoland/celld/blob/f2bf648663a610eefde71f3547ad61e9b896b1f0/crates/celld/ltx_repl.rs#L4514) +keeps ordered rounds in flight independently of bucket work; its +[follower stream](https://github.com/denoland/celld/blob/f2bf648663a610eefde71f3547ad61e9b896b1f0/crates/celld/node_log.rs#L160) +groups delivered frames before durable append. Cellule still awaits each append +batch. Shared storage components do not supply equivalent scheduling. + +Before, all submission phases reconcile with 27,617 successes and zero residual; +ordered-lock wait averages 115.63 ms, 98.33% of submission time. Candidate +observed ordered-wait mean is 118.81 ms. Its phase counts differ, 30,137–30,158, +with a 1,990,260,492-ns residual, so they cannot support an exact partition of +successful submission time. Candidate SQL-worker and follower-proof means are +0.209 and 23.742 ms in separate cohorts; these quantities are not additive or +CPU-utilization measurements. The ordering wait remains large, while the code +shows where publication credit enters that path. + +Candidate pending publications grow 2,025→2,465; unpublished node-log bytes +grow 13,300,853→34,648,517 and oldest debt grows 45.20→46.33 seconds. Retained +bytes fall 66,551,180→62,036,852 against 67,108,864. All 2,000 Cells remain +active and the node is unfenced with active Fleet at both boundaries. Issued / +follower-proven / tiered end positions are 52,662 / 52,572 / 52,036. +Two endpoints do not qualify a sustained bounded debt slope. Read-only windows +also include seed-root publication work. + +## Verification and next delivery + +All 13 broad contributor routes pass in the frozen source: 1,985 workspace +tests including doctests, 38 ignored; 60 local LTX tests; all-feature/target +checks; Clippy with warnings denied on Rust 1.97 and 1.99; format, API docs, +boundaries, layout, document, SQL/peer and harness gates. Focused results above +remain separate. The ignored environments remain unverified. + +Finish reducing catalog/base/history verification and upload work; separate +native progress from recoverably bounded publication debt; pipeline ordered +follower rounds with safe complete-suffix drain and recovery. Diagnose overload +availability and read variance. Larger queues alone cannot increase sustainable +service capacity. Complete Bucket adapter integration, mixed load, failed-owner +orchestration and safe collection. Keep the unchanged +[node capacity contract](../crates/cellule-runtime/docs/write-performance-design.md). + +Raw builds, source snapshots, invalid preparation attempts, failures, journals, +telemetry, audits and inventories stay outside Git under +`/Volumes/Workspace/crabbuild-target/cellule-write-perf-8ad1/metadata-windows-20261009-*`. +An early verification preparation race and a regular-file-only source comparison +are explicitly excluded/corrected; their records remain. Neither produced a +qualified performance result. All canonical reports and the comparison fail +qualification. + +[Preceding encoder measurement](pr67-encoder-cost-measurement.md), +[implementation](bundle-coverage-implementation.md), +[delivery](write-performance-delivery.md). diff --git a/docs/write-performance-delivery.md b/docs/write-performance-delivery.md index 96a53573..a33395e8 100644 --- a/docs/write-performance-delivery.md +++ b/docs/write-performance-delivery.md @@ -6,17 +6,18 @@ comparison retains exact retry and cold-state audits. **Celld write parity has not been established.** The [proposal](write-performance-proposal.md) remains the acceptance contract; completing tests or a load run does not pass its gates. -The latest [paired 2,000-Cell comparison](pr67-encoder-cost-measurement.md) -measures `30cd960` at **550.52 Fleet writes/s versus 512.02 before and 1,994.57 -for fresh celld**. Successful scheduled write p99 rises 853.65→866.06 ms; -returned errors increase and drops decrease. Read-only throughput is -17,877.42/s versus 13,780.35 before and celld's 19,522.95/s. All six cases pass -joined drain and complete ACK warm/cold read/retry audits. Publication cost -and debt remain, both read arms drop offers, and every performance profile -fails. This single short pair establishes no repeatable or attributable gain. -PR #67 remains a draft. The -[preceding catalog-overlap comparison](pr67-catalog-overlap-measurement.md) -and earlier failures below remain evidence. +The latest [paired 2,000-Cell comparison](pr67-metadata-window-measurement.md) +measures `4a5b001` at **502.28 Fleet writes/s versus 455.73 before and 1,999.83 +for fresh celld**. Observed GET/range work falls, but successful scheduled write +p99 worsens 827.14→1,454.42 ms and returned errors increase. Read-only throughput +falls 17,905.02→17,347.03/s; celld completes 19,983.30/s. The candidate passes all +48,888 ACK warm/cold reads and original retries and drains in 66.52 seconds. +The baseline write case fails two warm retries and never reaches cold recovery. +Both Cellule read arms drop offers, the read guardrail fails and every performance +profile fails. This single short pair establishes no acceptable, repeatable or +attributable gain. PR #67 remains a draft. The +[preceding encoder comparison](pr67-encoder-cost-measurement.md) and earlier +passing and failed evidence remain separate observations. The earlier [selection-readiness comparison](pr67-selection-readiness-measurement.md) measures `6c909a6` at **184.77 Fleet writes/s and 248.68 Bucket writes/s**, versus From 10370d20f52c0b2c6103b0df61c56a1252238d33 Mon Sep 17 00:00:00 2001 From: forhappy Date: Fri, 9 Oct 2026 08:26:42 -0700 Subject: [PATCH 075/102] docs: explain measured write publication bottleneck --- .../docs/write-performance-design.md | 9 ++ docs/pr67-publication-path-diagnosis.md | 141 ++++++++++++++++++ docs/write-performance-delivery.md | 8 + 3 files changed, 158 insertions(+) create mode 100644 docs/pr67-publication-path-diagnosis.md diff --git a/crates/cellule-runtime/docs/write-performance-design.md b/crates/cellule-runtime/docs/write-performance-design.md index ee19fbd1..b590ffaf 100644 --- a/crates/cellule-runtime/docs/write-performance-design.md +++ b/crates/cellule-runtime/docs/write-performance-design.md @@ -22,6 +22,15 @@ consumer expensive. Matching celld requires reducing that work and separating native progress from bounded recoverable publication debt; larger queues alone do not increase sustainable throughput. +The separate [fresh timing-only diagnosis](../../../docs/pr67-publication-path-diagnosis.md) +reproduces 468.42 writes/s versus 1,999.77 for celld. Serial selection averages +106.68 ms for 59.30 captures; checkpoint callbacks occupy another 14.83% of the +interval, explaining an approximate 473-capture/s service rate. Its submission +phases reconcile exactly: ordered wait averages 108.52 ms, 98.26% of submission +time. All 51,923 Cellule ACKs pass warm/cold reads and original retries, but +write errors/drops persist. This diagnostic adds timing only and establishes +no new production optimization or qualification. + Status: implementation in progress. The application path is not qualified at the targets below. Component I/O reductions are not application TPS. diff --git a/docs/pr67-publication-path-diagnosis.md b/docs/pr67-publication-path-diagnosis.md new file mode 100644 index 00000000..7cc85001 --- /dev/null +++ b/docs/pr67-publication-path-diagnosis.md @@ -0,0 +1,141 @@ +# PR 67: why write throughput still differs from celld + +**The fast paths are not architecturally equivalent yet.** Both systems use +SQLite WAL, LTX captures, fenced ownership, follower logs and bucket storage. +Cellule still couples native admission to a serial, expensive publication +consumer. That consumer sets the write rate under sustained pressure. +Performance parity remains unmet and PR #67 stays a draft. + +## What the code does + +Cellule's [assignment path](../crates/cellule-runtime/src/node/log_shipper/mod.rs) +acquires the global ordered lock, then awaits publication queue capacity before +committing a sequence and enqueueing follower work. The +[publication feed](../crates/cellule-runtime/src/node/log_shipper/publication/mod.rs) +has 512 submission slots and retains native-byte admission through selection. +The consumer selects one cohort of at most 64 captures/frames and 4 MiB before +starting the next. A full publication queue therefore blocks otherwise +independent Cells before their follower append can start. + +```mermaid +flowchart LR + SQL[SQLite commit and capture] --> Admission[Global ordered admission] + Admission --> Followers[Follower append] + Followers --> ACK[Fleet ACK] + Admission --> Publication[Serial bundle selection] + Publication -. Publication queue and byte credit .-> Admission +``` + +The reservation before issuance prevents cancellation from leaving a sequence +gap; bypassing it requires a replacement recoverable, bounded ownership path. +Moving the wait or enlarging the queue alone cannot raise sustainable service. + +Celld's [shipping loop](https://github.com/denoland/celld/blob/f2bf648663a610eefde71f3547ad61e9b896b1f0/crates/celld/ltx_repl.rs#L4530) +keeps ordered rounds in flight independently of bucket work. Its +[follower stream](https://github.com/denoland/celld/blob/f2bf648663a610eefde71f3547ad61e9b896b1f0/crates/celld/node_log.rs#L179) +groups already delivered requests into one durable append. Cellule awaits each +append batch, and its example HTTP adapter holds a member grant mutex across +the RPC. Concurrent futures alone would not establish safe delivery ordering. + +Publication also does different work. Celld's +[bundle flush](https://github.com/denoland/celld/blob/f2bf648663a610eefde71f3547ad61e9b896b1f0/crates/celld/node_log.rs#L6621) +uploads new native entries, then reads the original log record to check that +the epoch remains open before crediting the upload. Cellule reloads selected +catalog metadata, builds and uploads a proposal, freshly verifies its origin, +base roots and historical chains, and selects a new head by authority CAS. +These are different publication protocols and service costs. + +## Fresh timing-only reproduction + +The external probe pins production `4a5b00148ba579c1f4e035b16e8eb783c3d6fe6d` +and adds operation-local timing to six files. Scheduling, proofs, formats and +budgets are unchanged. All exported framework bytes match the preceding pinned +build; client, auditor, fixtures and images also match. Instrumentation can +perturb timing, so this is diagnosis, not a new production optimization result. + +Both fresh cases use 2,000 uniformly offered real Cells, 96-byte SQL values, +the same two-hour request/result ledger, one owner/two followers, tmpfs, +128 clients/queue slots, 30-second warmup and a 60-second 2,000-write/s window. +Both measured SQLite paths use WAL NORMAL. No build, contributor suite or +journal audit overlaps a timed window. + +| Diagnostic | Successful writes/s | Successful scheduled p99 ms | Errors | Dropped offers | +| --- | ---: | ---: | ---: | ---: | +| Cellule, timing instrumented | 468.42 | 549.23 | 49,665 | 42,167 | +| celld reference | 1,999.77 | 29.38 | 0 | 0 | + +Every Cell has timed successful responses. Independent replay reconciles all +attempts, outputs, errors, dropped offers and ACK manifests. All 51,923 Cellule +ACKs and 182,001 celld ACKs pass warm/cold reads and original retries. Joined +drain takes 77.40 and 18.84 seconds respectively. Cellule warmup also has +9,538 errors and 28,708 drops; celld has neither. These passing ACK audits do +not turn the overloaded Cellule performance profile into a pass. + +### Measured service, not a SQLite throughput ceiling + +| Operation-local phase | Fully contained operations | Mean ms | +| --- | ---: | ---: | +| Complete authority selection | 474 | 106.68 | +| Preparation within those selections | 474 | 50.61 | +| Verification/selection within those selections | 474 | 54.78 | +| Preparation catalog load | 474 | 16.45 | +| Proposal encoding | 474 | 8.21 | +| Proposal PUT | 474 | 25.36 | +| Base verification | 475 | 24.36 | +| Historical chain verification | 475 | 14.15 | +| Complete checkpoint callback | 365 | 24.37 | + +The 474 complete selections cover 28,110 captures: 59.30 per cohort. They +consume 50.57 seconds of the 59.987-second metrics interval; checkpoint +callbacks consume another 8.895 seconds, about 14.83% of the interval. +The source runs these authority operations serially. Their approximate service +rate is `28,110 / (50.5672 + 8.8950) = 472.74 captures/s`, close to the observed +468.42 successful writes/s. This explains the order of magnitude, not maximum +capacity. Producer receipt-credit timing includes some checkpoint work and +must not be added again. Nested phase cohorts have unequal boundary counts; +do not sum every row in the table. + +All submission phases have the same 28,173-success cohort and zero timing +residual. Ordered-lock wait averages **108.52 ms**, 98.26% of submission time; +publication-slot wait while holding the lock averages 1.80 ms. SQL worker +execution averages 0.205 ms, capture 0.141 ms and follower-proof wait 34.90 ms +in separate cohorts. These timers are not additive and do not measure CPU +utilization. The dominant submission wait and serial publication service +support prioritizing that dependency over SQLite execution. + +### Interpreting the older numbers + +The [older 144.50 versus 4,622.43 result](pr67-bounded-history-measurement.md) +used 1,000 Cells and 15,000 offered writes/s. Celld dropped 622,398 offers and +did not complete cold/drain qualification after an OOM during shutdown. +It establishes neither sustainable 4,622-write/s capacity nor a comparison +with the current 2,000-Cell profile. The latest +[uninstrumented production pair](pr67-metadata-window-measurement.md) remains +502.28 versus 1,999.83 writes/s, with Cellule latency/error and read regressions. +The offered-load cap prevents either 2K reference from proving celld's maximum. + +## Required architectural delivery + +1. Separate follower progress from publication waits with byte-accounted, + recoverable capture ownership, preserving complete issued-range drain. +2. Use ordered member lanes and bounded rounds in flight, with follower group + commit and ordered proof application. Preserve reconnect, cancellation, + fencing and unresolved-prefix behavior. +3. Make publication incremental and pipeline expensive immutable work around + ordered selection. Keep exact original scope/range proofs, authenticated + lookup, safe checkpoint ordering and bounded debt. A single 64-capture + consumer needs less than 32 ms/cohort to sustain 2K/s, before checkpoint + overhead; the current probe measures 106.68 ms for about 59 captures. +4. Verify sustained write/read/mixed capacity, overload availability, complete + failed-owner recovery and collection against the unchanged + [capacity contract](../crates/cellule-runtime/docs/write-performance-design.md). + +The shared Docker VM has eight CPUs and about 8 GiB total across all roles; +internal resource policies differ between systems. These short cases do not +qualify a dedicated 8-vCPU/16-GiB serving process or physical-media durability. +Every canonical report fails qualification. No production code changes or new +read/Bucket measurements were made in this diagnostic turn. + +Raw sources, timing overlays, binaries, journals, audits and phase/model data +remain outside Git under +`/Volumes/Workspace/crabbuild-target/cellule-write-perf-8ad1/publication-path-20261009-*`. diff --git a/docs/write-performance-delivery.md b/docs/write-performance-delivery.md index a33395e8..d55837fe 100644 --- a/docs/write-performance-delivery.md +++ b/docs/write-performance-delivery.md @@ -19,6 +19,14 @@ attributable gain. PR #67 remains a draft. The [preceding encoder comparison](pr67-encoder-cost-measurement.md) and earlier passing and failed evidence remain separate observations. +The separate [publication-path diagnosis](pr67-publication-path-diagnosis.md) +adds external timing only to the same production revision. It reproduces +468.42 writes/s versus 1,999.77 for celld and reconciles complete warm/cold +ACK audits. Serial selection takes 106.68 ms per 59.30 captures, with checkpoint +work consuming another 14.83% of the interval. Ordered-lock wait averages +108.52 ms in an exact submission partition. Write errors/drops remain; this +supports the bottleneck mechanism and makes no new production gain claim. + The earlier [selection-readiness comparison](pr67-selection-readiness-measurement.md) measures `6c909a6` at **184.77 Fleet writes/s and 248.68 Bucket writes/s**, versus 195.13 and 247.55 before. Its delayed-selection actor regression passes, but From 2b6db11a21ef81e906a46b092e172f326840faab Mon Sep 17 00:00:00 2001 From: forhappy Date: Fri, 9 Oct 2026 09:01:01 -0700 Subject: [PATCH 076/102] Keep bundle base verification slots busy --- .../docs/write-performance-design.md | 10 +- .../src/node/bundle/tests/faults.rs | 8 ++ .../node/bundle/tests/index/base_cohort.rs | 97 +++++++++++++++++++ .../src/node/bundle/verification/base.rs | 64 +++++------- 4 files changed, 138 insertions(+), 41 deletions(-) diff --git a/crates/cellule-runtime/docs/write-performance-design.md b/crates/cellule-runtime/docs/write-performance-design.md index b590ffaf..51c1d2db 100644 --- a/crates/cellule-runtime/docs/write-performance-design.md +++ b/crates/cellule-runtime/docs/write-performance-design.md @@ -100,9 +100,13 @@ verification. After the fresh body matches, selection shares the proposal allocation and uses its released buffer allowance for 2 MiB of historical scratch and at most 2 MiB of compact facts/planning metadata. Eight bounded reads overlap; every frame and original Cell chain remains canonically checked. Fresh small -packed leaf bases now overlap in a separate phase: eight one-use canonical -origin plans charge at most 512 KiB each, reusing the same 4-MiB allowance after -the complete cohort body matches. All base jobs join or drop before historical +packed leaf bases overlap in a separate phase through eight continuous slots. +Each slot retains its fresh root read through complete canonical dependency +verification, then immediately starts the next base; a slow root does not hold +completed slots behind a group barrier. One-use origin plans charge at most +512 KiB per slot, reusing the same 4-MiB allowance after the complete cohort +body matches. All bounded slots join or drop before larger serial graphs run, +and all base jobs join or drop before historical scratch/facts admission. Larger base graphs and individual historical extents above 2 MiB retain serial verification. Selection retains no reconstruction bodies beyond the operation. Working admission remains 20 MiB; workload diff --git a/crates/cellule-runtime/src/node/bundle/tests/faults.rs b/crates/cellule-runtime/src/node/bundle/tests/faults.rs index 5dd851a0..22d7b869 100644 --- a/crates/cellule-runtime/src/node/bundle/tests/faults.rs +++ b/crates/cellule-runtime/src/node/bundle/tests/faults.rs @@ -143,6 +143,14 @@ impl ObjectStore for ReplyFault { } if (mode == 12 && path.as_ref().ends_with(".root")) || (mode == 13 && path.as_ref().ends_with(".pack")) + || (mode == 16 + && path.as_ref().ends_with(".root") + && self + .held_metadata + .lock() + .unwrap() + .iter() + .any(|(object, _)| object == path.as_ref())) { self.base_started.fetch_add(1, Ordering::SeqCst); self.node_started.notify_one(); diff --git a/crates/cellule-runtime/src/node/bundle/tests/index/base_cohort.rs b/crates/cellule-runtime/src/node/bundle/tests/index/base_cohort.rs index 1cbff0e5..2301b1f1 100644 --- a/crates/cellule-runtime/src/node/bundle/tests/index/base_cohort.rs +++ b/crates/cellule-runtime/src/node/bundle/tests/index/base_cohort.rs @@ -11,6 +11,103 @@ async fn fresh_base_body_reads_overlap_without_selecting_before_verification() { held_bases(13).await; } +#[tokio::test] +async fn completed_base_verification_refills_slots_while_one_root_read_is_held() { + let faults = Arc::new(super::super::faults::ReplyFault::default()); + let mut f = Fixture::with_store(faults.clone()).await; + super::super::coverage::enroll(&mut f).await; + let mut cells = Vec::new(); + for number in 4..14 { + cells.push(f.cell(number).await); + f.heartbeat().await; + } + let mut frames = Vec::new(); + let mut assigned = Vec::new(); + for cell in &mut cells { + let (_, capture, range) = f.append(cell, 2); + frames.extend(capture); + assigned.push(range); + } + let now = f.node.advertisement().issued_at_ms(); + let proposal = f + .directory + .prepare_node_bundle(&f.node, &frames, &assigned, now) + .await + .unwrap(); + // Hold a member of the first scheduling group, rather than relying on + // Cell creation order to match the catalog's authenticated digest order. + let base = proposal.catalog.bindings[0].control.ltx_root().unwrap(); + let path = f.layout.incarnation_object_path( + &base.cell, + &base.incarnation, + &base.digest, + cellule_ltx::CellObjectKind::Root, + ); + faults + .held_metadata + .lock() + .unwrap() + .push((path.to_string(), 0)); + f.count.reset(); + faults.mode.store(16, Ordering::SeqCst); + { + let selection = + f.directory + .select_node_bundle(&f.node, &proposal, &f.lease, Limits::default(), now); + tokio::pin!(selection); + let progress = async { + loop { + let roots = f + .count + .requests() + .into_iter() + .filter(|request| request.location.ends_with(".root")) + .count(); + if roots > 8 { + break; + } + tokio::task::yield_now().await; + } + }; + tokio::select! { + _ = &mut selection => panic!("unverified held root must prevent authority selection"), + result = tokio::time::timeout(std::time::Duration::from_secs(1), progress) => { + assert!(result.is_ok(), "completed base verification must refill a slot before the held root finishes"); + }, + } + assert_eq!(faults.base_started.load(Ordering::SeqCst), 1); + assert!( + f.count + .requests() + .iter() + .any(|request| request.location.ends_with(".pack")), + "refill follows real dependency verification, not just root prefetch" + ); + } + assert_eq!( + f.count.put_requests(), + 0, + "cancellation selects no authority" + ); + faults.mode.store(0, Ordering::SeqCst); + faults.node_resume.notify_waiters(); + f.count.reset(); + let (_, proofs) = f + .directory + .select_node_bundle(&f.node, &proposal, &f.lease, Limits::default(), now) + .await + .unwrap(); + assert_eq!(proofs.len(), cells.len()); + assert!(proofs.iter().all(|proof| proof.commit_sequence() == 2)); + let roots = f + .count + .requests() + .into_iter() + .filter(|request| request.location.ends_with(".root")) + .count(); + assert_eq!(roots, cells.len(), "retry freshly verifies every base root"); +} + async fn held_bases(mode: u8) { let faults = Arc::new(super::super::faults::ReplyFault::default()); let mut f = Fixture::with_store(faults.clone()).await; diff --git a/crates/cellule-runtime/src/node/bundle/verification/base.rs b/crates/cellule-runtime/src/node/bundle/verification/base.rs index 52a3595c..c147183f 100644 --- a/crates/cellule-runtime/src/node/bundle/verification/base.rs +++ b/crates/cellule-runtime/src/node/bundle/verification/base.rs @@ -1,4 +1,4 @@ -//! Admitted fresh small-base verification; larger graphs stay serial. +//! Continuous admitted small-base verification; larger graphs stay serial. use super::*; pub(super) async fn verify( @@ -6,45 +6,33 @@ pub(super) async fn verify( bindings: &[Binding], limits: cellule_ltx::Limits, ) -> Result<()> { - for cohort in bindings.chunks(READ_CONCURRENCY) { - let mut small = Vec::with_capacity(cohort.len()); - let mut serial = Vec::with_capacity(cohort.len()); - let mut plans = stream::iter(0..cohort.len()) - .map(|index| async move { - Ok::<_, Error>(( - index, - proof::prepare_base_origin(layout, &cohort[index], limits).await?, - )) - }) - .buffer_unordered(READ_CONCURRENCY); - while let Some(plan) = plans.next().await { - let (index, plan): (_, Option) = plan?; - match plan { - Some(plan) => small.push(plan), - None => serial.push(index), + let mut serial = Vec::with_capacity(bindings.len()); + // Each slot owns the root read through complete dependency verification. + // Refilling a completed slot avoids waiting for an unrelated slow root; + // eight original 512-KiB charges still fit the released 4-MiB allowance. + let mut reads = stream::iter(0..bindings.len()) + .map(|index| async move { + let Some(plan) = proof::prepare_base_origin(layout, &bindings[index], limits).await? + else { + return Ok::<_, Error>(Some(index)); + }; + if plan.working_bytes() > MAX_BUNDLE_BYTES as usize / READ_CONCURRENCY { + return Err(Error::Capacity("bundle base verification bytes")); } + plan.verify().await?; + Ok(None) + }) + .buffer_unordered(READ_CONCURRENCY); + while let Some(result) = reads.next().await { + if let Some(index) = result? { + serial.push(index); } - drop(plans); - let working = small.iter().try_fold(0_usize, |bytes, plan| { - bytes - .checked_add(plan.working_bytes()) - .ok_or(Error::Capacity("bundle base verification bytes")) - })?; - if working > MAX_BUNDLE_BYTES as usize { - return Err(Error::Capacity("bundle base verification bytes")); - } - let mut reads = stream::iter(small) - .map(|plan| async move { plan.verify().await.map(|_| ()) }) - .buffer_unordered(READ_CONCURRENCY); - while let Some(result) = reads.next().await { - result?; - } - drop(reads); - // All bounded operations have joined/dropped before a larger graph - // takes the original serial working set. No partial plan grants CAS. - for index in serial { - proof::verify_base(layout, &cohort[index], limits).await?; - } + } + drop(reads); + // All bounded operations have joined/dropped before a larger graph + // takes the original serial working set. No partial plan grants CAS. + for index in serial { + proof::verify_base(layout, &bindings[index], limits).await?; } Ok(()) } From fa79e5471910f16059a7abfde991c75e5bd1de26 Mon Sep 17 00:00:00 2001 From: forhappy Date: Fri, 9 Oct 2026 09:25:50 -0700 Subject: [PATCH 077/102] Revert "Keep bundle base verification slots busy" This reverts commit 2b6db11a21ef81e906a46b092e172f326840faab. --- .../docs/write-performance-design.md | 10 +- .../src/node/bundle/tests/faults.rs | 8 -- .../node/bundle/tests/index/base_cohort.rs | 97 ------------------- .../src/node/bundle/verification/base.rs | 64 +++++++----- 4 files changed, 41 insertions(+), 138 deletions(-) diff --git a/crates/cellule-runtime/docs/write-performance-design.md b/crates/cellule-runtime/docs/write-performance-design.md index 51c1d2db..b590ffaf 100644 --- a/crates/cellule-runtime/docs/write-performance-design.md +++ b/crates/cellule-runtime/docs/write-performance-design.md @@ -100,13 +100,9 @@ verification. After the fresh body matches, selection shares the proposal allocation and uses its released buffer allowance for 2 MiB of historical scratch and at most 2 MiB of compact facts/planning metadata. Eight bounded reads overlap; every frame and original Cell chain remains canonically checked. Fresh small -packed leaf bases overlap in a separate phase through eight continuous slots. -Each slot retains its fresh root read through complete canonical dependency -verification, then immediately starts the next base; a slow root does not hold -completed slots behind a group barrier. One-use origin plans charge at most -512 KiB per slot, reusing the same 4-MiB allowance after the complete cohort -body matches. All bounded slots join or drop before larger serial graphs run, -and all base jobs join or drop before historical +packed leaf bases now overlap in a separate phase: eight one-use canonical +origin plans charge at most 512 KiB each, reusing the same 4-MiB allowance after +the complete cohort body matches. All base jobs join or drop before historical scratch/facts admission. Larger base graphs and individual historical extents above 2 MiB retain serial verification. Selection retains no reconstruction bodies beyond the operation. Working admission remains 20 MiB; workload diff --git a/crates/cellule-runtime/src/node/bundle/tests/faults.rs b/crates/cellule-runtime/src/node/bundle/tests/faults.rs index 22d7b869..5dd851a0 100644 --- a/crates/cellule-runtime/src/node/bundle/tests/faults.rs +++ b/crates/cellule-runtime/src/node/bundle/tests/faults.rs @@ -143,14 +143,6 @@ impl ObjectStore for ReplyFault { } if (mode == 12 && path.as_ref().ends_with(".root")) || (mode == 13 && path.as_ref().ends_with(".pack")) - || (mode == 16 - && path.as_ref().ends_with(".root") - && self - .held_metadata - .lock() - .unwrap() - .iter() - .any(|(object, _)| object == path.as_ref())) { self.base_started.fetch_add(1, Ordering::SeqCst); self.node_started.notify_one(); diff --git a/crates/cellule-runtime/src/node/bundle/tests/index/base_cohort.rs b/crates/cellule-runtime/src/node/bundle/tests/index/base_cohort.rs index 2301b1f1..1cbff0e5 100644 --- a/crates/cellule-runtime/src/node/bundle/tests/index/base_cohort.rs +++ b/crates/cellule-runtime/src/node/bundle/tests/index/base_cohort.rs @@ -11,103 +11,6 @@ async fn fresh_base_body_reads_overlap_without_selecting_before_verification() { held_bases(13).await; } -#[tokio::test] -async fn completed_base_verification_refills_slots_while_one_root_read_is_held() { - let faults = Arc::new(super::super::faults::ReplyFault::default()); - let mut f = Fixture::with_store(faults.clone()).await; - super::super::coverage::enroll(&mut f).await; - let mut cells = Vec::new(); - for number in 4..14 { - cells.push(f.cell(number).await); - f.heartbeat().await; - } - let mut frames = Vec::new(); - let mut assigned = Vec::new(); - for cell in &mut cells { - let (_, capture, range) = f.append(cell, 2); - frames.extend(capture); - assigned.push(range); - } - let now = f.node.advertisement().issued_at_ms(); - let proposal = f - .directory - .prepare_node_bundle(&f.node, &frames, &assigned, now) - .await - .unwrap(); - // Hold a member of the first scheduling group, rather than relying on - // Cell creation order to match the catalog's authenticated digest order. - let base = proposal.catalog.bindings[0].control.ltx_root().unwrap(); - let path = f.layout.incarnation_object_path( - &base.cell, - &base.incarnation, - &base.digest, - cellule_ltx::CellObjectKind::Root, - ); - faults - .held_metadata - .lock() - .unwrap() - .push((path.to_string(), 0)); - f.count.reset(); - faults.mode.store(16, Ordering::SeqCst); - { - let selection = - f.directory - .select_node_bundle(&f.node, &proposal, &f.lease, Limits::default(), now); - tokio::pin!(selection); - let progress = async { - loop { - let roots = f - .count - .requests() - .into_iter() - .filter(|request| request.location.ends_with(".root")) - .count(); - if roots > 8 { - break; - } - tokio::task::yield_now().await; - } - }; - tokio::select! { - _ = &mut selection => panic!("unverified held root must prevent authority selection"), - result = tokio::time::timeout(std::time::Duration::from_secs(1), progress) => { - assert!(result.is_ok(), "completed base verification must refill a slot before the held root finishes"); - }, - } - assert_eq!(faults.base_started.load(Ordering::SeqCst), 1); - assert!( - f.count - .requests() - .iter() - .any(|request| request.location.ends_with(".pack")), - "refill follows real dependency verification, not just root prefetch" - ); - } - assert_eq!( - f.count.put_requests(), - 0, - "cancellation selects no authority" - ); - faults.mode.store(0, Ordering::SeqCst); - faults.node_resume.notify_waiters(); - f.count.reset(); - let (_, proofs) = f - .directory - .select_node_bundle(&f.node, &proposal, &f.lease, Limits::default(), now) - .await - .unwrap(); - assert_eq!(proofs.len(), cells.len()); - assert!(proofs.iter().all(|proof| proof.commit_sequence() == 2)); - let roots = f - .count - .requests() - .into_iter() - .filter(|request| request.location.ends_with(".root")) - .count(); - assert_eq!(roots, cells.len(), "retry freshly verifies every base root"); -} - async fn held_bases(mode: u8) { let faults = Arc::new(super::super::faults::ReplyFault::default()); let mut f = Fixture::with_store(faults.clone()).await; diff --git a/crates/cellule-runtime/src/node/bundle/verification/base.rs b/crates/cellule-runtime/src/node/bundle/verification/base.rs index c147183f..52a3595c 100644 --- a/crates/cellule-runtime/src/node/bundle/verification/base.rs +++ b/crates/cellule-runtime/src/node/bundle/verification/base.rs @@ -1,4 +1,4 @@ -//! Continuous admitted small-base verification; larger graphs stay serial. +//! Admitted fresh small-base verification; larger graphs stay serial. use super::*; pub(super) async fn verify( @@ -6,33 +6,45 @@ pub(super) async fn verify( bindings: &[Binding], limits: cellule_ltx::Limits, ) -> Result<()> { - let mut serial = Vec::with_capacity(bindings.len()); - // Each slot owns the root read through complete dependency verification. - // Refilling a completed slot avoids waiting for an unrelated slow root; - // eight original 512-KiB charges still fit the released 4-MiB allowance. - let mut reads = stream::iter(0..bindings.len()) - .map(|index| async move { - let Some(plan) = proof::prepare_base_origin(layout, &bindings[index], limits).await? - else { - return Ok::<_, Error>(Some(index)); - }; - if plan.working_bytes() > MAX_BUNDLE_BYTES as usize / READ_CONCURRENCY { - return Err(Error::Capacity("bundle base verification bytes")); + for cohort in bindings.chunks(READ_CONCURRENCY) { + let mut small = Vec::with_capacity(cohort.len()); + let mut serial = Vec::with_capacity(cohort.len()); + let mut plans = stream::iter(0..cohort.len()) + .map(|index| async move { + Ok::<_, Error>(( + index, + proof::prepare_base_origin(layout, &cohort[index], limits).await?, + )) + }) + .buffer_unordered(READ_CONCURRENCY); + while let Some(plan) = plans.next().await { + let (index, plan): (_, Option) = plan?; + match plan { + Some(plan) => small.push(plan), + None => serial.push(index), } - plan.verify().await?; - Ok(None) - }) - .buffer_unordered(READ_CONCURRENCY); - while let Some(result) = reads.next().await { - if let Some(index) = result? { - serial.push(index); } - } - drop(reads); - // All bounded operations have joined/dropped before a larger graph - // takes the original serial working set. No partial plan grants CAS. - for index in serial { - proof::verify_base(layout, &bindings[index], limits).await?; + drop(plans); + let working = small.iter().try_fold(0_usize, |bytes, plan| { + bytes + .checked_add(plan.working_bytes()) + .ok_or(Error::Capacity("bundle base verification bytes")) + })?; + if working > MAX_BUNDLE_BYTES as usize { + return Err(Error::Capacity("bundle base verification bytes")); + } + let mut reads = stream::iter(small) + .map(|plan| async move { plan.verify().await.map(|_| ()) }) + .buffer_unordered(READ_CONCURRENCY); + while let Some(result) = reads.next().await { + result?; + } + drop(reads); + // All bounded operations have joined/dropped before a larger graph + // takes the original serial working set. No partial plan grants CAS. + for index in serial { + proof::verify_base(layout, &cohort[index], limits).await?; + } } Ok(()) } From 9b79fd2498f13929232e9d27ea787be0c909f3d0 Mon Sep 17 00:00:00 2001 From: forhappy Date: Fri, 9 Oct 2026 09:38:42 -0700 Subject: [PATCH 078/102] Record failed base-slot trial and fresh write comparison --- .../docs/write-performance-design.md | 10 +- docs/pr67-base-pipeline-measurement.md | 96 +++++++++++++++++++ docs/pr67-publication-path-diagnosis.md | 5 +- docs/write-performance-delivery.md | 9 +- 4 files changed, 117 insertions(+), 3 deletions(-) create mode 100644 docs/pr67-base-pipeline-measurement.md diff --git a/crates/cellule-runtime/docs/write-performance-design.md b/crates/cellule-runtime/docs/write-performance-design.md index b590ffaf..32617882 100644 --- a/crates/cellule-runtime/docs/write-performance-design.md +++ b/crates/cellule-runtime/docs/write-performance-design.md @@ -1,6 +1,14 @@ # Node write and read performance design -The [current paired 2,000-Cell measurements](../../../docs/pr67-metadata-window-measurement.md) +The [latest fresh write comparison](../../../docs/pr67-base-pipeline-measurement.md) +completes 579.10 Fleet writes/s for unchanged production code, 570.25/s for a +continuous-base-slot trial and 1,999.83/s for celld. The trial has no acceptable +TPS/latency gain and is reverted. Both Cellule cases fail warm ACK availability +checks with HTTP 503s and never reach cold audit; celld passes all 182,001 warm +and cold mutations and original retries. These short shared-VM runs establish +no qualified capacity. The acceptance contract below is unchanged. + +The [preceding paired 2,000-Cell measurements](../../../docs/pr67-metadata-window-measurement.md) complete 502.28 Fleet writes/s versus 455.73 before and 1,999.83 for fresh celld. Metadata windows reduce observed GET/range work, but successful scheduled write p99 worsens 827.14→1,454.42 ms and returned errors increase. Read-only diff --git a/docs/pr67-base-pipeline-measurement.md b/docs/pr67-base-pipeline-measurement.md new file mode 100644 index 00000000..269b2815 --- /dev/null +++ b/docs/pr67-base-pipeline-measurement.md @@ -0,0 +1,96 @@ +# PR 67: continuous base verification did not improve write TPS + +**The trial is reverted; performance parity remains unmet.** The fresh unchanged +build completes 579.10 Fleet writes/s, the continuous-slot candidate 570.25/s, +and celld 1,999.83/s at the same 2,000/s offered load. Candidate throughput is +1.53% lower and successful scheduled p99 is 6.61% higher in this single pair. +Both Cellule cases fail warm ACK availability checks and never reach cold +recovery. This establishes no acceptable or attributable performance gain. +PR #67 remains a draft. + +## Change and verification + +The unchanged framework revision is `10370d20f52c0b2c6103b0df61c56a1252238d33`; +its production code is the preceding `4a5b001` implementation. Experimental +revision `2b6db11a21ef81e906a46b092e172f326840faab` replaces groups of eight +base reads with eight continuously refilled slots. Each slot retains one fresh +root through complete dependency verification; larger graphs remain serial. +The 4-MiB base allowance and 20-MiB producer admission are unchanged. + +The real SQLite regression holds one first-group root, verifies later bases +can progress, cancels before authority selection and retries with fresh reads. +It fails three times on the unchanged verifier and passes three times on the +candidate. All 92 tests in the focused native bundle inventory pass. The +isolated candidate passes all 13 contributor routes: 1,986 workspace tests, +60 local LTX tests and both Rust 1.97/1.99 clippy checks. The 38 environment +dependent workspace tests remain ignored, not qualified. + +A copied native Cargo cache initially reused the baseline binary. That attempt +is retained as invalid; distinct source and identical binary hashes expose it. +Forcing the changed crate to rebuild produces a different binary and the +passing results above. A controller's historical expected count of 100 was +applied to the wrong scope: `node::bundle::tests::` selects 92 tests, +while `node::bundle::` selects 100. The full workspace suite passes all 100, +including the eight additional index/proof tests. No tests or qualification +thresholds are dropped. + +The trial is reverted because the application measurement supplies no +performance benefit. Its commit, sources, regression and failed measurements +remain available for review. Production source after the revert matches the +unchanged revision exactly. + +## Fresh matched diagnostic + +Both Linux release builds use the canonical pinned builder. The driver, +auditor, fixture bytes, images and loaded runner match; only four intended +framework files differ. Celld is pinned to +`f2bf648663a610eefde71f3547ad61e9b896b1f0`. Runs execute sequentially, with +2,000 uniform Cells, 96-byte SQL-ledger values, 128 clients/queue slots, +30-second warmup and a 60-second Fleet window. Both SQLite paths use WAL NORMAL. +No build, contributor check or independent journal audit overlaps a timed window. + +| Arm | Successful writes/s | Successful scheduled p99 ms | Errors | Dropped offers | +| --- | ---: | ---: | ---: | ---: | +| Cellule unchanged | 579.10 | 416.19 | 54,145 | 30,984 | +| Cellule trial, reverted | 570.25 | 443.71 | 49,849 | 35,866 | +| celld | 1,999.83 | 35.24 | 0 | 0 | + +The p99 cohort includes successful measured offers through client drain; +fast failed attempts do not reduce it. Warmup errors/drops are respectively +19,116/18,264 for the unchanged build, 7,560/26,503 for the trial and 0/0 for +celld. Every Cell has successful measured responses in every arm. + +Independent replay reconciles every client attempt, original successful output, +in-window/trailing completion, per-Cell count, payload counter, dropped offer, +loaded identity and complete ACK stream. ACK counts are 59,492, 62,223 and +182,001. Cellule warm audits encounter 467 and 2,779 HTTP 503s respectively; +neither reaches cold audit. These are availability failures, not evidence of +lost data. Celld verifies all 182,001 mutations and original retries in both +warm and cold audits, and drains the original fleet in 16.55 seconds. + +All canonical qualification reports fail. The shared VM has eight CPUs and +8,306,286,592 bytes of total memory across all roles; container ceilings exceed +it. Cellule's original 64-MiB retained and 1-GiB disk policies have no equivalent +celld fixture settings. These cases match workload and container ceilings, not +effective internal admission. A short overload observation is not dedicated +8-vCPU/16-GiB capacity, maximum celld throughput or physical-media durability. +Failed Cellule audits also leave required later lifecycle evidence incomplete. + +## Architectural implication + +The [publication diagnosis](pr67-publication-path-diagnosis.md) remains the +mechanism: publication capacity blocks global issuance before follower work, +while serial selection repeatedly verifies bases and history and competes with +checkpoints. The trial improves a component's scheduling under a held read but +does not remove this dependency or establish application TPS improvement. +Delivery must separate bounded recoverable native progress from publication, +pipeline ordered follower lanes with group commit, reduce repeated publication +work and preserve ACK read/retry availability under pressure. The +[capacity contract](../crates/cellule-runtime/docs/write-performance-design.md) +is unchanged; write/read/mixed and full recovery qualification remain open. + +Fresh snapshots, builds, logs, journals, independent audits and the invalid +attempt records remain outside Git under +`/Volumes/Workspace/crabbuild-target/base-pipeline-20261009-independent`. +Earlier raw diagnostic directories disappeared; their committed reports remain +historical records, and missing raw evidence is not claimed to be reverified. diff --git a/docs/pr67-publication-path-diagnosis.md b/docs/pr67-publication-path-diagnosis.md index 7cc85001..48260718 100644 --- a/docs/pr67-publication-path-diagnosis.md +++ b/docs/pr67-publication-path-diagnosis.md @@ -137,5 +137,8 @@ Every canonical report fails qualification. No production code changes or new read/Bucket measurements were made in this diagnostic turn. Raw sources, timing overlays, binaries, journals, audits and phase/model data -remain outside Git under +were recorded outside Git under `/Volumes/Workspace/crabbuild-target/cellule-write-perf-8ad1/publication-path-20261009-*`. +That external directory later disappeared; its missing raw evidence is not +claimed to be reverified. The [fresh write comparison](pr67-base-pipeline-measurement.md) +retains newly rebuilt sources, verification and journals in a separate directory. diff --git a/docs/write-performance-delivery.md b/docs/write-performance-delivery.md index d55837fe..8ebf1700 100644 --- a/docs/write-performance-delivery.md +++ b/docs/write-performance-delivery.md @@ -6,7 +6,14 @@ comparison retains exact retry and cold-state audits. **Celld write parity has not been established.** The [proposal](write-performance-proposal.md) remains the acceptance contract; completing tests or a load run does not pass its gates. -The latest [paired 2,000-Cell comparison](pr67-metadata-window-measurement.md) +The latest [fresh write comparison](pr67-base-pipeline-measurement.md) completes +579.10 Fleet writes/s for unchanged production code, 570.25/s for the trial and +1,999.83/s for celld. The trial is reverted: this pair establishes no acceptable +TPS/latency improvement. Cellule's warm ACK audits fail with 467 and 2,779 HTTP +503s; neither reaches cold audit. Celld passes all 182,001 warm/cold mutations +and original retries. All profiles remain unqualified and PR #67 stays a draft. + +The preceding [paired 2,000-Cell comparison](pr67-metadata-window-measurement.md) measures `4a5b001` at **502.28 Fleet writes/s versus 455.73 before and 1,999.83 for fresh celld**. Observed GET/range work falls, but successful scheduled write p99 worsens 827.14→1,454.42 ms and returned errors increase. Read-only throughput From 01de6e12de48858024cb23c28c6c954321d07706 Mon Sep 17 00:00:00 2001 From: forhappy Date: Fri, 9 Oct 2026 10:12:23 -0700 Subject: [PATCH 079/102] Verify publication pressure against retained write telemetry --- docs/pr67-publication-path-diagnosis.md | 46 +++++++++++++++++++++++-- 1 file changed, 43 insertions(+), 3 deletions(-) diff --git a/docs/pr67-publication-path-diagnosis.md b/docs/pr67-publication-path-diagnosis.md index 48260718..fadcf1d3 100644 --- a/docs/pr67-publication-path-diagnosis.md +++ b/docs/pr67-publication-path-diagnosis.md @@ -45,6 +45,43 @@ catalog metadata, builds and uploads a proposal, freshly verifies its origin, base roots and historical chains, and selects a new head by authority CAS. These are different publication protocols and service costs. +## Cross-check against the latest retained run + +Re-analysis of the unchanged `10370d2` run in the +[latest matched comparison](pr67-base-pipeline-measurement.md) verifies the +start/end telemetry files against the frozen evidence index. This is an +analysis of that existing 60-second run, not a new benchmark or optimization. + +All seven native submission phase deltas have the same **34,809** completed +submission cohort. Their total durations add exactly to the complete timer: + +| Native submission phase | Mean ms | +| --- | ---: | +| Complete submission | 96.669 | +| Waiting for the global ordered lane | 95.297 | +| Waiting for publication capacity while holding that lane | 1.267 | +| Local capture load/verification | 0.096 | +| Sequence assignment and enqueue | 0.008 | + +The ordered-lane wait accounts for **98.58%** of native submission duration. +These are overlapping callers' queue waits, not a serial service interval; +inverting 95.297 ms would not yield node TPS. The code dependency explains why +publication backpressure becomes a queue shared by otherwise independent Cells. + +Separate cohorts record a 0.187-ms mean SQL worker duration (88,882 executions), +0.140-ms capture duration (34,871 captures), and 21.543-ms follower-proof wait +(34,809 observations). Those means must not be added to the submission phases +or treated as successful-command SQL costs. Follower data-sync means are below +0.001 ms on tmpfs; this supplies no physical-media fsync comparison. Per-Cell +publication debt rises from 2,051 to 2,514 pending publications; its oldest age +is already about 45 seconds at the start of the window. + +The underlying result remains **579.10 successful writes/s**, with errors, +drops and failed warm availability checks. Re-analysis does not improve it or +qualify capacity. The checked input hashes, exact count/duration partitions and +replay script are retained outside Git under +`/Volumes/Workspace/crabbuild-target/write-path-explanation-20261009`. + ## Fresh timing-only reproduction The external probe pins production `4a5b00148ba579c1f4e035b16e8eb783c3d6fe6d` @@ -110,9 +147,12 @@ used 1,000 Cells and 15,000 offered writes/s. Celld dropped 622,398 offers and did not complete cold/drain qualification after an OOM during shutdown. It establishes neither sustainable 4,622-write/s capacity nor a comparison with the current 2,000-Cell profile. The latest -[uninstrumented production pair](pr67-metadata-window-measurement.md) remains -502.28 versus 1,999.83 writes/s, with Cellule latency/error and read regressions. -The offered-load cap prevents either 2K reference from proving celld's maximum. +[uninstrumented production pair](pr67-base-pipeline-measurement.md) records +579.10 versus 1,999.83 writes/s, with Cellule errors, drops and failed warm ACK +availability. The offered-load cap prevents either 2K reference from proving +celld's maximum. The earlier +[metadata-window pair](pr67-metadata-window-measurement.md) recorded 502.28 +versus 1,999.83 writes/s, with Cellule latency/error and read regressions. ## Required architectural delivery From c384039909b0348f83714d065557a1b84ac0d0f8 Mon Sep 17 00:00:00 2001 From: forhappy Date: Fri, 9 Oct 2026 10:13:23 -0700 Subject: [PATCH 080/102] Correct pinned celld bundle flush source anchor --- docs/pr67-publication-path-diagnosis.md | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/docs/pr67-publication-path-diagnosis.md b/docs/pr67-publication-path-diagnosis.md index fadcf1d3..4a402c23 100644 --- a/docs/pr67-publication-path-diagnosis.md +++ b/docs/pr67-publication-path-diagnosis.md @@ -38,7 +38,7 @@ append batch, and its example HTTP adapter holds a member grant mutex across the RPC. Concurrent futures alone would not establish safe delivery ordering. Publication also does different work. Celld's -[bundle flush](https://github.com/denoland/celld/blob/f2bf648663a610eefde71f3547ad61e9b896b1f0/crates/celld/node_log.rs#L6621) +[bundle flush](https://github.com/denoland/celld/blob/f2bf648663a610eefde71f3547ad61e9b896b1f0/crates/celld/node_log.rs#L6436) uploads new native entries, then reads the original log record to check that the epoch remains open before crediting the upload. Cellule reloads selected catalog metadata, builds and uploads a proposal, freshly verifies its origin, From 6e4dc5917716f35e7e246dc887fa40816e21c13c Mon Sep 17 00:00:00 2001 From: forhappy Date: Fri, 9 Oct 2026 10:53:36 -0700 Subject: [PATCH 081/102] Decouple native admission and pipeline ordered follower appends --- .../docs/failover-and-followers.md | 32 +- .../docs/write-performance-design.md | 12 + .../src/node/log_shipper/mod.rs | 269 ++++++++------- .../src/node/log_shipper/publication/mod.rs | 52 ++- .../src/node/log_shipper/replication/mod.rs | 213 ++++++++++++ .../src/node/log_shipper/tests.rs | 306 +++++++++++++++--- 6 files changed, 692 insertions(+), 192 deletions(-) create mode 100644 crates/cellule-runtime/src/node/log_shipper/replication/mod.rs diff --git a/crates/cellule-runtime/docs/failover-and-followers.md b/crates/cellule-runtime/docs/failover-and-followers.md index 0230d15f..ce2585c9 100644 --- a/crates/cellule-runtime/docs/failover-and-followers.md +++ b/crates/cellule-runtime/docs/failover-and-followers.md @@ -282,11 +282,15 @@ set, activation bit, contiguous object watermark, and renewable recovery claim. **Shipper.** `NodeLogShipper`: -- Reserves encoded bytes before assigning a sequence. +- Reserves capture bytes and bounded queue bookkeeping before assigning a sequence. - Multiplexes accepted cuts in submission order. - Batches for at most one millisecond or 64 frames. -- Sends each batch to every member concurrently. -- Advances the gate only after all receipts cover the batch. +- Enqueues each batch synchronously into every member's ordered FIFO, with at + most eight original rounds in flight. Member I/O progresses independently. +- Groups already queued adjacent requests into one canonical follower append, + preserving the 64-frame and native-byte limits and safe coverage watermark. +- Applies credits in original round order, only after all receipts cover that + round. A grouped higher watermark cannot skip an unresolved earlier round. - Stops fleet issuance for that epoch on any encoding, transport, or receipt failure, while its tickets remain eligible for object proof and covered rotation. @@ -294,17 +298,23 @@ set, activation bit, contiguous object watermark, and renewable recovery claim. `NodeDurability::take_publication_feed` installs one ordered consumer before the original epoch issues any frames. Each `AssignedCapture` retains its exact complete assignment and verified frames under the shipper's existing byte -admission. The 512-submission queue reserves space before sequence issuance; -cancellation or a full queue cannot leave an issued gap. A capture larger than +admission. Publication has no separate submission-slot wait under the global +ordering lock. Its FIFO and member I/O retain the original native credit; +exhausted byte credit blocks before issuance. Capture control allocations and +member frame-vector copies consume that same original window. A capture larger than 64 frames remains one complete witness even when follower transport splits it. -Shutdown closes admission, wakes blocked producers and lets the feed drain -accepted captures. The host must join selection or verified fallback for all +Shutdown closes admission, wakes blocked producers and joins every accepted +member append. Cancelling a shutdown waiter does not permit a later join to +return before the original worker and member tasks finish. The feed can drain +accepted captures; the host must join selection or verified fallback for all of them before retiring the epoch. -This feed grants no publication authority or response proof. Ordinary actor -bundle ACKs remain disabled until shared selection, visibility, capture release -and complete issued-range drain are integrated. The example application does -not install the feed yet; this API alone is not a measured throughput gain. +This feed grants no publication authority or response proof. The Fleet SQL +example installs the managed producer before activation. Actor bundle responses +require its original dependency-verified selection receipt and exact capture +match; absent or late coverage uses the canonical root path. The Bucket-only +performance fixture bypasses this producer. Native pipeline tests alone do not +establish application throughput, sustained publication capacity or qualification. **Ensemble directory.** It filters live peers by protocol, pressure, and the exact shared-disk capacity advertised by their follower stores: diff --git a/crates/cellule-runtime/docs/write-performance-design.md b/crates/cellule-runtime/docs/write-performance-design.md index 32617882..a1b16827 100644 --- a/crates/cellule-runtime/docs/write-performance-design.md +++ b/crates/cellule-runtime/docs/write-performance-design.md @@ -72,6 +72,18 @@ capacity contract does not turn a failed earlier profile into a passing one. ## Canonical write path +The current native candidate removes publication-slot waits from global +issuance. Capture bodies, control allocations and member frame-vector copies +consume the original native byte window before a ticket commits; publication +and accepted member I/O keep that credit through their original lifetimes. +Eight original rounds can overlap in per-member FIFO lanes, with ordered credit +application and bounded grouping into canonical follower append/fsync. Every +lane joins before epoch shutdown, including after cancellation of a join waiter. +This changes native execution; the bundle publisher still runs expensive +selection cohorts serially. Publication staging, reduced repeated verification +and complete application performance qualification remain required. Component +passes establish no new TPS claim. + ```mermaid flowchart LR A[Bounded admission before SQL] --> B[Mutation and retry result commit together] diff --git a/crates/cellule-runtime/src/node/log_shipper/mod.rs b/crates/cellule-runtime/src/node/log_shipper/mod.rs index a0d86fb5..e69c0acb 100644 --- a/crates/cellule-runtime/src/node/log_shipper/mod.rs +++ b/crates/cellule-runtime/src/node/log_shipper/mod.rs @@ -2,7 +2,7 @@ use std::{collections::VecDeque, io::Read as _, sync::Arc, time::Duration}; use bytes::Bytes; -use futures_util::future::join_all; +use futures_util::stream::StreamExt; use tokio::sync::{OwnedSemaphorePermit, Semaphore, mpsc}; use crate::identity::{ApplicationId, CellId}; @@ -12,6 +12,7 @@ use crate::node::log_transport::{AppendRequest, NodeLogTransport}; use crate::{Error, Result}; mod publication; +mod replication; mod submission; pub use publication::{AssignedCapture, NodePublicationFeed, SelectedBundlePublication}; pub(crate) use publication::{SelectedBundle, SubmittedCapture}; @@ -20,6 +21,7 @@ const MAX_BATCH_FRAMES: usize = 64; const MAX_QUEUED_SUBMISSIONS: usize = 512; const NODE_FRAME_HEADER_BYTES: u64 = 240; const BATCH_INTERVAL: Duration = Duration::from_millis(1); +type WorkerResult = std::result::Result<(), Arc>; /// One captured Cell commit awaiting ordered node-log assignment. pub struct NodeLogSubmission { @@ -194,14 +196,16 @@ fn read_segment(path: &std::path::Path, expected_bytes: u64, limit: u64) -> Resu /// after every selected member returns an fsynced contiguous watermark. pub struct NodeLogShipper { sender: std::sync::Mutex>>, - worker: std::sync::Mutex>>, + worker: std::sync::Mutex>>, bytes: Arc, order: tokio::sync::Mutex<()>, max_outstanding_bytes: u64, + member_count: usize, gate: DurabilityGate, limits: cellule_ltx::Limits, publication: publication::PublicationState, stopping: tokio::sync::watch::Sender, + terminal: tokio::sync::watch::Receiver>, telemetry: crate::fleet::telemetry::CellTelemetryHandle, } @@ -245,29 +249,42 @@ impl NodeLogShipper { let bytes = Arc::new(Semaphore::new(permits)); let worker_gate = gate.clone(); let (stopping, _) = tokio::sync::watch::channel(false); - let worker = runtime.spawn(run_shipper( - receiver, - worker_gate, - Arc::clone(&bytes), - transport, - leader, - log_epoch, - members, - batch_bytes, - telemetry.clone(), - interval, - stopping.clone(), - )); + let member_count = members.len(); + let (completed, terminal) = tokio::sync::watch::channel(None); + let worker_bytes = Arc::clone(&bytes); + let worker_telemetry = telemetry.clone(); + let worker_stopping = stopping.clone(); + let worker = runtime.spawn(async move { + let result = run_shipper( + receiver, + worker_gate, + worker_bytes, + transport, + leader, + log_epoch, + members, + batch_bytes, + worker_telemetry, + interval, + worker_stopping, + ) + .await + .map_err(Arc::new); + completed.send_replace(Some(result.clone())); + result + }); Ok(Self { sender: std::sync::Mutex::new(Some(sender)), worker: std::sync::Mutex::new(Some(worker)), bytes, order: tokio::sync::Mutex::new(()), max_outstanding_bytes: batch_bytes, + member_count, gate, limits, publication: publication::PublicationState::default(), stopping, + terminal, telemetry, }) } @@ -354,7 +371,12 @@ impl NodeLogShipper { { return Err(Error::Capacity("node-log submission")); } - let permit_count = u32::try_from(submission.encoded_bytes) + let retained_bytes = + publication::retained_bytes(submission.encoded_bytes, frame_count, self.member_count)?; + if retained_bytes > self.max_outstanding_bytes { + return Err(Error::Capacity("node-log retained capture bytes")); + } + let permit_count = u32::try_from(retained_bytes) .ok() .filter(|bytes| *bytes != 0) .ok_or(Error::Capacity("node-log outstanding bytes"))?; @@ -386,10 +408,11 @@ impl NodeLogShipper { // atomically commits their consecutive ticket before enqueueing. observation.enter(Stage::OrderedLane); let _ordered = self.order.lock().await; - // Reserve both consumers before committing a sequence. Cancellation or - // a full publication queue therefore cannot leave an unselectable gap. + // Native credit owns both consumers' complete capture until they join. + // Checking the publication consumer is synchronous: storage capacity + // never waits under this global ordered lane. observation.enter(Stage::PublicationSlot); - let publication = self.publication.reserve(&self.stopping).await?; + let publication = self.publication.sender()?; if *self.stopping.borrow() { return Err(Error::RuntimeClosed); } @@ -406,7 +429,12 @@ impl NodeLogShipper { let selection = if let Some(publication) = publication { let (capture, selection) = AssignedCapture::new(assignment, encoded.clone(), Arc::clone(&reservation)); - publication.send(capture); + if publication.send(capture).is_err() { + // A lost consumer cannot permit later Fleet acknowledgements + // beyond this unpublished assignment. Fence before shipping. + stop_shipper(&self.gate, &self.bytes, &self.stopping); + return Err(Error::RuntimeClosed); + } Some(selection) } else { None @@ -442,8 +470,19 @@ impl NodeLogShipper { .map_err(|_| Error::Node("node-log shipper lock poisoned"))? .take(); let joined = match worker { - Some(worker) => worker.await.map_err(Error::FollowerWorkerJoin), - None => Ok(()), + Some(worker) => worker + .await + .map_err(Error::FollowerWorkerJoin)? + .map_err(Error::Shared), + None => { + let mut terminal = self.terminal.clone(); + loop { + if let Some(result) = terminal.borrow_and_update().clone() { + break result.map_err(Error::Shared); + } + terminal.changed().await.map_err(|_| Error::RuntimeClosed)?; + } + } }; self.gate.stop_shipping(); publication_closed.and(joined) @@ -477,7 +516,7 @@ struct QueuedFrame { reason = "the worker keeps the exact log epoch, ensemble and bounded admission explicit" )] async fn run_shipper( - mut receiver: mpsc::Receiver, + receiver: mpsc::Receiver, gate: DurabilityGate, bytes: Arc, transport: Arc, @@ -488,26 +527,76 @@ async fn run_shipper( telemetry: crate::fleet::telemetry::CellTelemetryHandle, interval: Duration, stopping: tokio::sync::watch::Sender, +) -> Result<()> { + let lanes = + replication::MemberLanes::start(transport, leader, log_epoch, members, max_batch_bytes); + run_rounds( + receiver, + &gate, + &bytes, + &lanes, + max_batch_bytes, + &telemetry, + interval, + &stopping, + ) + .await; + // Closing the round waiter never cancels a member's accepted native I/O. + // Join every original lane before shutdown can release the epoch. + let result = lanes.join().await; + if result.is_err() { + stop_shipper(&gate, &bytes, &stopping); + } + result +} + +async fn run_rounds( + mut receiver: mpsc::Receiver, + gate: &DurabilityGate, + bytes: &Semaphore, + lanes: &replication::MemberLanes, + max_batch_bytes: u64, + telemetry: &crate::fleet::telemetry::CellTelemetryHandle, + interval: Duration, + stopping: &tokio::sync::watch::Sender, ) { let mut pending = VecDeque::::new(); let mut closed = false; + let mut rounds = futures_util::stream::FuturesOrdered::new(); loop { - if pending.is_empty() { - if closed { - bytes.close(); - stopping.send_replace(true); + if closed && pending.is_empty() && rounds.is_empty() { + bytes.close(); + stopping.send_replace(true); + return; + } + if rounds.len() == replication::PIPELINE || (closed && pending.is_empty()) { + if let Some(round) = rounds.next().await + && !complete_round(gate, telemetry, round) + { + stop_shipper(gate, bytes, stopping); + receiver.close(); return; } - match receiver.recv().await { - Some(submission) => pending.extend(submission.frames), - None => { - bytes.close(); - stopping.send_replace(true); - return; + continue; + } + if pending.is_empty() { + tokio::select! { + round = rounds.next(), if !rounds.is_empty() => { + if let Some(round) = round + && !complete_round(gate, telemetry, round) + { + stop_shipper(gate, bytes, stopping); + receiver.close(); + return; + } + continue; + } + submission = receiver.recv() => match submission { + Some(submission) => pending.extend(submission.frames), + None => { closed = true; continue; } } } } - let deadline = tokio::time::Instant::now() + interval; let mut batch = Vec::::new(); let mut batch_bytes = 0_u64; @@ -517,18 +606,18 @@ async fn run_shipper( break; }; let Some(next_bytes) = batch_bytes.checked_add(next.encoded.len() as u64) else { - stop_shipper(&gate, &bytes, &stopping); + stop_shipper(gate, bytes, stopping); return; }; if !batch.is_empty() && next_bytes > max_batch_bytes { break; } if next_bytes > max_batch_bytes { - stop_shipper(&gate, &bytes, &stopping); + stop_shipper(gate, bytes, stopping); return; } let Some(next) = pending.pop_front() else { - stop_shipper(&gate, &bytes, &stopping); + stop_shipper(gate, bytes, stopping); return; }; batch_bytes = next_bytes; @@ -550,32 +639,34 @@ async fn run_shipper( Err(_) => break, } } - - let append_bytes = batch - .iter() - .try_fold(0_u64, |total, frame| { - total.checked_add(frame.encoded.len() as u64) - }) - .and_then(|bytes| bytes.checked_mul(members.len() as u64)) - .unwrap_or(u64::MAX); - let result = append_batch( - &gate, - Arc::clone(&transport), - leader, - log_epoch, - &members, - batch, - ) - .await; - telemetry.node_log_append(result.is_ok(), append_bytes); - if result.is_err() { - stop_shipper(&gate, &bytes, &stopping); - receiver.close(); - return; + // Enqueue synchronously, in original sequence order, before returning a + // future. Poll order can never reorder a member's accepted append lane. + match lanes.enqueue(batch, gate.tiered_through()) { + Ok(round) => rounds.push_back(round), + Err(_) => { + stop_shipper(gate, bytes, stopping); + receiver.close(); + return; + } } } } +fn complete_round( + gate: &DurabilityGate, + telemetry: &crate::fleet::telemetry::CellTelemetryHandle, + (bytes, result): replication::RoundResult, +) -> bool { + let result = result.and_then(|acknowledgements| { + for (member, through) in acknowledgements { + gate.acknowledge(member, through)?; + } + Ok(()) + }); + telemetry.node_log_append(result.is_ok(), bytes); + result.is_ok() +} + fn stop_shipper( gate: &DurabilityGate, bytes: &Semaphore, @@ -586,65 +677,5 @@ fn stop_shipper( bytes.close(); } -async fn append_batch( - gate: &DurabilityGate, - transport: Arc, - leader: crate::SessionId, - log_epoch: u64, - members: &[NodeId], - batch: Vec, -) -> Result<()> { - let first = batch - .first() - .ok_or(Error::Node("node-log append batch is empty"))? - .sequence; - let last = batch - .last() - .ok_or(Error::Node("node-log append batch is empty"))? - .sequence; - if !batch - .windows(2) - .all(|pair| pair[0].sequence.checked_add(1) == Some(pair[1].sequence)) - { - return Err(Error::Node("node-log append batch is not contiguous")); - } - let frames = batch - .iter() - .map(|frame| frame.encoded.clone()) - .collect::>(); - let covered_through = gate.tiered_through(); - let replies = join_all(members.iter().map(|member| { - let transport = Arc::clone(&transport); - let request = AppendRequest { - leader_session: leader, - log_epoch, - frames: frames.clone(), - covered_through, - }; - let member = *member; - async move { (member, transport.append(member, request).await) } - })) - .await; - let mut acknowledgements = Vec::with_capacity(replies.len()); - for (member, reply) in replies { - let receipt = reply?; - // A queued batch can include an already object-covered prefix. Only - // its uncovered suffix must remain on the follower for a fleet proof. - let required_first = first.max(covered_through.saturating_add(1)); - if receipt.durable_through < last - || receipt.base_sequence == 0 - || receipt.base_sequence > receipt.durable_through.saturating_add(1) - || (last > covered_through && receipt.base_sequence > required_first) - { - return Err(Error::Node("node-log append receipt differs")); - } - acknowledgements.push((member, receipt.durable_through)); - } - for (member, durable_through) in acknowledgements { - gate.acknowledge(member, durable_through)?; - } - Ok(()) -} - #[cfg(test)] mod tests; diff --git a/crates/cellule-runtime/src/node/log_shipper/publication/mod.rs b/crates/cellule-runtime/src/node/log_shipper/publication/mod.rs index c38deb2a..0b8400ad 100644 --- a/crates/cellule-runtime/src/node/log_shipper/publication/mod.rs +++ b/crates/cellule-runtime/src/node/log_shipper/publication/mod.rs @@ -109,12 +109,13 @@ impl AssignedCapture { /// Sole ordered consumer for this original native epoch's publication work. /// -/// Complete captures share the shipper's outstanding-byte limit and a queue of -/// at most 512 submissions. Receiving does not release admission: the consumer +/// Complete captures and their bookkeeping share the shipper's outstanding-byte +/// limit. Receiving does not release admission: the consumer /// retains each value through joined selection or verified object fallback. -/// A slow consumer applies backpressure before new sequence issuance. +/// Publication has no independent slot wait in the global issuance lane; +/// exhausted native byte credit applies backpressure before that lane. pub struct NodePublicationFeed { - receiver: mpsc::Receiver, + receiver: mpsc::UnboundedReceiver, stopping: watch::Receiver, } @@ -142,22 +143,23 @@ impl NodePublicationFeed { #[derive(Default)] pub(super) struct PublicationState { - sender: OnceLock>>>, + sender: OnceLock>>>, } impl PublicationState { pub(super) fn take_feed(&self, stopping: watch::Receiver) -> Result { - let (sender, receiver) = mpsc::channel(MAX_QUEUED_SUBMISSIONS); + // Every queued value owns charged native credit, including control and + // frame-vector bookkeeping. The byte window bounds this FIFO even when + // the origin publisher is held; a second slot budget would couple Fleet + // progress to bucket latency again. + let (sender, receiver) = mpsc::unbounded_channel(); self.sender .set(Mutex::new(Some(sender))) .map_err(|_| Error::Node("node publication feed already installed"))?; Ok(NodePublicationFeed { receiver, stopping }) } - pub(super) async fn reserve( - &self, - stopping: &watch::Sender, - ) -> Result>> { + pub(super) fn sender(&self) -> Result>> { let Some(sender) = self.sender.get() else { return Ok(None); }; @@ -166,14 +168,10 @@ impl PublicationState { .map_err(|_| Error::Node("node publication feed lock poisoned"))? .clone() .ok_or(Error::RuntimeClosed)?; - let mut stopping = stopping.subscribe(); - if *stopping.borrow() { + if sender.is_closed() { return Err(Error::RuntimeClosed); } - tokio::select! { - permit = sender.reserve_owned() => permit.map(Some).map_err(|_| Error::RuntimeClosed), - _ = stopping.changed() => Err(Error::RuntimeClosed), - } + Ok(Some(sender)) } pub(super) fn close(&self) -> Result<()> { @@ -186,3 +184,25 @@ impl PublicationState { Ok(()) } } + +/// Body and bounded capture/queue control allocations consume the same original +/// byte window. The fixed allowance covers watch/Arc/channel bookkeeping; +/// frame vectors are charged by their concrete element sizes. No independent +/// count limit or larger native window is introduced. +pub(super) fn retained_bytes(body: u64, frames: u64, members: usize) -> Result { + let member_vectors = members + .checked_mul(std::mem::size_of::() + std::mem::size_of::>()) + .ok_or(Error::Capacity("node-log retained member bytes"))?; + let control = (members as u64) + .checked_mul(128) + .and_then(|bytes| bytes.checked_add(1024)) + .ok_or(Error::Capacity("node-log retained member bytes"))?; + let per_frame = 2 * std::mem::size_of::() + + std::mem::size_of::() + + member_vectors; + frames + .checked_mul(per_frame as u64) + .and_then(|metadata| metadata.checked_add(control)) + .and_then(|metadata| body.checked_add(metadata)) + .ok_or(Error::Capacity("node-log retained capture bytes")) +} diff --git a/crates/cellule-runtime/src/node/log_shipper/replication/mod.rs b/crates/cellule-runtime/src/node/log_shipper/replication/mod.rs new file mode 100644 index 00000000..9ad22ec4 --- /dev/null +++ b/crates/cellule-runtime/src/node/log_shipper/replication/mod.rs @@ -0,0 +1,213 @@ +//! Ordered member lanes, bounded rounds, and group commit of queued native work. +use super::*; +use crate::follower::FollowerReceipt; +use futures_util::future::{BoxFuture, join_all}; +use tokio::sync::oneshot; + +pub(super) const PIPELINE: usize = 8; +pub(super) type RoundResult = (u64, Result>); +type Round = BoxFuture<'static, RoundResult>; + +struct MemberAppend { + first: u64, + last: u64, + frames: Vec, + covered_through: u64, + // Member I/O keeps native credit even when its round waiter is cancelled. + _retention: Vec>, + completed: oneshot::Sender>, +} + +pub(super) struct MemberLanes { + senders: Vec<(NodeId, mpsc::Sender)>, + workers: Vec>, +} + +impl MemberLanes { + pub(super) fn start( + transport: Arc, + leader: crate::SessionId, + epoch: u64, + members: Vec, + max_bytes: u64, + ) -> Self { + let mut senders = Vec::with_capacity(members.len()); + let mut workers = Vec::with_capacity(members.len()); + for member in members { + let (sender, receiver) = mpsc::channel(PIPELINE); + senders.push((member, sender)); + workers.push(tokio::spawn(run_member( + Arc::clone(&transport), + member, + leader, + epoch, + max_bytes, + receiver, + ))); + } + Self { senders, workers } + } + + pub(super) fn enqueue(&self, batch: Vec, covered: u64) -> Result { + let first = batch + .first() + .ok_or(Error::Node("empty append round"))? + .sequence; + let last = batch + .last() + .ok_or(Error::Node("empty append round"))? + .sequence; + if !batch + .windows(2) + .all(|pair| pair[0].sequence.checked_add(1) == Some(pair[1].sequence)) + { + return Err(Error::Node("node-log append batch is not contiguous")); + } + let bytes = batch + .iter() + .try_fold(0_u64, |total, frame| { + total.checked_add(frame.encoded.len() as u64) + }) + .and_then(|bytes| bytes.checked_mul(self.senders.len() as u64)) + .ok_or(Error::Capacity("append round byte count"))?; + let mut replies = Vec::with_capacity(self.senders.len()); + for (member, sender) in &self.senders { + let (completed, reply) = oneshot::channel(); + let request = MemberAppend { + first, + last, + covered_through: covered, + frames: batch.iter().map(|frame| frame.encoded.clone()).collect(), + _retention: batch + .iter() + .map(|frame| Arc::clone(&frame._reservation)) + .collect(), + completed, + }; + // The global eight-round window bounds each member's FIFO. No + // future is polled here, so request delivery preserves submission order. + sender.try_send(request).map_err(|_| Error::RuntimeClosed)?; + replies.push((*member, reply)); + } + Ok(Box::pin(async move { + let replies = join_all(replies.into_iter().map(|(member, reply)| async move { + ( + member, + reply + .await + .map_err(|_| Error::RuntimeClosed) + .and_then(|result| result), + ) + })) + .await; + // Keep the original round's credit until every member answer joins. + let _retention = batch; + let mut acknowledgements = Vec::with_capacity(replies.len()); + let result = (|| { + for (member, reply) in replies { + let receipt = reply?; + let required_first = first.max(covered.saturating_add(1)); + if receipt.durable_through < last + || receipt.base_sequence == 0 + || receipt.base_sequence > receipt.durable_through.saturating_add(1) + || (last > covered && receipt.base_sequence > required_first) + { + return Err(Error::Node("node-log append receipt differs")); + } + // A grouped receipt can cover later queued rounds. Credit + // only this original round; unresolved predecessors cannot be skipped. + acknowledgements.push((member, last)); + } + Ok(acknowledgements) + })(); + (bytes, result) + })) + } + + pub(super) async fn join(self) -> Result<()> { + drop(self.senders); + let joined = join_all(self.workers).await; + for result in joined { + result.map_err(Error::FollowerWorkerJoin)?; + } + Ok(()) + } +} + +async fn run_member( + transport: Arc, + member: NodeId, + leader: crate::SessionId, + epoch: u64, + max_bytes: u64, + mut receiver: mpsc::Receiver, +) { + let mut carry = None; + loop { + let first = match carry.take() { + Some(first) => first, + None => match receiver.recv().await { + Some(first) => first, + None => return, + }, + }; + let mut frame_count = first.frames.len(); + let mut bytes = first + .frames + .iter() + .map(|frame| frame.len() as u64) + .sum::(); + let mut last = first.last; + let mut covered = first.covered_through; + let mut requests = vec![first]; + while frame_count < MAX_BATCH_FRAMES { + let Ok(next) = receiver.try_recv() else { + break; + }; + let next_bytes = next + .frames + .iter() + .map(|frame| frame.len() as u64) + .sum::(); + if last.checked_add(1) != Some(next.first) + || frame_count + next.frames.len() > MAX_BATCH_FRAMES + || bytes + .checked_add(next_bytes) + .is_none_or(|total| total > max_bytes) + { + carry = Some(next); + break; + } + frame_count += next.frames.len(); + bytes += next_bytes; + last = next.last; + covered = covered.min(next.covered_through); + requests.push(next); + } + let frames = requests + .iter_mut() + .flat_map(|request| std::mem::take(&mut request.frames)) + .collect(); + let result = transport + .append( + member, + AppendRequest { + leader_session: leader, + log_epoch: epoch, + frames, + covered_through: covered, + }, + ) + .await + .map_err(Arc::new); + // One canonical follower append covers the delivered group and its + // single fsync. Every original request receives the same verified watermark. + for request in requests { + let result = result + .as_ref() + .copied() + .map_err(|source| Error::Shared(Arc::clone(source))); + let _ = request.completed.send(result); + } + } +} diff --git a/crates/cellule-runtime/src/node/log_shipper/tests.rs b/crates/cellule-runtime/src/node/log_shipper/tests.rs index 9d632242..7d493e37 100644 --- a/crates/cellule-runtime/src/node/log_shipper/tests.rs +++ b/crates/cellule-runtime/src/node/log_shipper/tests.rs @@ -14,6 +14,7 @@ struct RecordingTransport { fail: Option, delay: Option, receipt: Option, + held: Option<(NodeId, Arc, Arc)>, } #[derive(Default)] @@ -83,6 +84,7 @@ impl RecordingTransport { fail: Some(member), delay: None, receipt: None, + held: None, } } @@ -92,6 +94,7 @@ impl RecordingTransport { fail: None, delay: Some(delay), receipt: None, + held: None, } } @@ -113,6 +116,16 @@ impl NodeLogTransport for RecordingTransport { request: AppendRequest, ) -> BoxFuture<'a, Result> { Box::pin(async move { + if let Some((held, entered, release)) = &self.held + && member == *held + && request.frames.first().is_some_and(|frame| { + cellule_ltx::inspect_node_frame(frame.clone(), cellule_ltx::Limits::default()) + .is_ok_and(|frame| frame.scope().node_sequence == 1) + }) + { + entered.notify_one(); + release.notified().await; + } if self.fail == Some(member) { return Err(Error::Node("injected follower failure")); } @@ -356,11 +369,16 @@ async fn splits_large_submission_at_sixty_four_frames() { ); shipper.shutdown().await.unwrap(); assert!(publication.recv().await.is_none()); - let retained = capture - .frames() - .iter() - .map(|frame| frame.encoded().len()) - .sum::(); + let retained = publication::retained_bytes( + capture + .frames() + .iter() + .map(|frame| frame.encoded().len() as u64) + .sum(), + capture.frames().len() as u64, + 1, + ) + .unwrap() as usize; assert_eq!( shipper.bytes.available_permits(), shipper.max_outstanding_bytes as usize - retained @@ -392,42 +410,99 @@ fn publication_submission(cuts: &cellule_ltx::CaptureBatch, index: u64) -> NodeL } #[tokio::test] -async fn cancelled_full_publication_queue_does_not_issue_a_native_gap() { +async fn publication_backlog_does_not_block_follower_issuance_with_byte_credit() { + let (_directory, cuts) = capture(); + let gate = DurabilityGate::new(session(1), node(1), 2, [node(2)]).unwrap(); + gate.activate_fleet().unwrap(); + let shipper = NodeLogShipper::new( + gate.clone(), + Arc::new(RecordingTransport::default()), + cellule_ltx::Limits::default(), + ) + .unwrap(); + let mut feed = shipper.take_publication_feed().unwrap(); + for index in 0..512 { + let ticket = shipper + .submit(publication_submission(&cuts, index)) + .await + .unwrap(); + gate.wait_followers(ticket).await.unwrap(); + } + assert!(shipper.bytes.available_permits() > submission(&cuts).encoded_bytes as usize); + let next = tokio::time::timeout( + Duration::from_millis(100), + shipper.submit(publication_submission(&cuts, 512)), + ) + .await; + if let Ok(Ok(ticket)) = next.as_ref() { + gate.wait_followers(*ticket).await.unwrap(); + } + shipper.shutdown().await.unwrap(); + let mut accepted = 0; + while let Some(capture) = feed.recv().await { + accepted += 1; + assert_eq!(capture.assignment().ticket().first_sequence(), accepted); + capture.assignment().verify(capture.frames()).unwrap(); + } + assert!( + matches!(next, Ok(Ok(_))), + "available native byte credit must let followers advance while publication is paused" + ); + assert_eq!(accepted, 513); + assert_eq!( + shipper.bytes.available_permits(), + shipper.max_outstanding_bytes as usize + ); +} + +#[tokio::test] +async fn cancelled_full_native_byte_window_does_not_issue_a_native_gap() { let (_directory, cuts) = capture(); let gate = DurabilityGate::new(session(1), node(1), 2, [node(2)]).unwrap(); gate.activate_fleet().unwrap(); let telemetry = Arc::new(RecordingTelemetry::default()); + let limits = cellule_ltx::Limits { + max_capture_bytes: cuts.segments[0].info().size_bytes * 16, + ..cellule_ltx::Limits::default() + }; + let charge = publication::retained_bytes( + submission(&cuts).encoded_bytes, + cuts.segments.len() as u64, + 1, + ) + .unwrap(); + let accepted = NodeLogShipper::validate_limits(limits).unwrap().0 / charge; let shipper = NodeLogShipper::new_with_telemetry( gate.clone(), Arc::new(RecordingTransport::default()), - cellule_ltx::Limits::default(), + limits, crate::fleet::telemetry::CellTelemetryHandle::from_sink(telemetry.clone()), ) .unwrap(); let mut feed = shipper.take_publication_feed().unwrap(); assert!(shipper.take_publication_feed().is_err()); - for index in 0..MAX_QUEUED_SUBMISSIONS as u64 { + for index in 0..accepted { shipper .submit(publication_submission(&cuts, index)) .await .unwrap(); } - assert_eq!(gate.issued_through(), 512); - let blocked = shipper.submit(publication_submission(&cuts, 512)); + assert_eq!(gate.issued_through(), accepted); + let blocked = shipper.submit(publication_submission(&cuts, accepted)); assert!( tokio::time::timeout(Duration::from_millis(25), blocked) .await .is_err() ); - assert_eq!(gate.issued_through(), 512); + assert_eq!(gate.issued_through(), accepted); { let observed = telemetry.submissions.lock().unwrap(); - assert_eq!(observed.len(), 513); + assert_eq!(observed.len() as u64, accepted + 1); let (cell, blocked) = observed.last().unwrap(); - assert_eq!(*cell, publication_submission(&cuts, 512).cell); + assert_eq!(*cell, publication_submission(&cuts, accepted).cell); assert!(blocked.cancelled); assert!(!blocked.succeeded); - assert!(blocked.publication_slot > Duration::ZERO); + assert!(blocked.native_bytes > Duration::ZERO); for (_, timing) in observed.iter() { assert_eq!( timing.validation @@ -449,13 +524,13 @@ async fn cancelled_full_publication_queue_does_not_issue_a_native_gap() { first.assignment().verify(first.frames()).unwrap(); drop(first); let next = shipper - .submit(publication_submission(&cuts, 512)) + .submit(publication_submission(&cuts, accepted)) .await .unwrap(); - assert_eq!(next.first_sequence(), 513); + assert_eq!(next.first_sequence(), accepted + 1); gate.wait_followers(next).await.unwrap(); shipper.shutdown().await.unwrap(); - for expected in 2..=513 { + for expected in 2..=accepted + 1 { let capture = feed.recv().await.unwrap(); assert_eq!(capture.assignment().ticket().first_sequence(), expected); capture.assignment().verify(capture.frames()).unwrap(); @@ -468,50 +543,52 @@ async fn cancelled_full_publication_queue_does_not_issue_a_native_gap() { } #[tokio::test] -async fn shutdown_wakes_full_publication_admission_and_joins_accepted_frames() { +async fn shutdown_wakes_full_native_byte_admission_and_joins_accepted_frames() { let (_directory, cuts) = capture(); let gate = DurabilityGate::new(session(1), node(1), 2, [node(2)]).unwrap(); let transport = Arc::new(RecordingTransport::default()); - let shipper = Arc::new( - NodeLogShipper::new( - gate.clone(), - transport.clone(), - cellule_ltx::Limits::default(), - ) - .unwrap(), - ); + let limits = cellule_ltx::Limits { + max_capture_bytes: cuts.segments[0].info().size_bytes * 16, + ..cellule_ltx::Limits::default() + }; + let charge = publication::retained_bytes( + submission(&cuts).encoded_bytes, + cuts.segments.len() as u64, + 1, + ) + .unwrap(); + let accepted = NodeLogShipper::validate_limits(limits).unwrap().0 / charge; + let shipper = Arc::new(NodeLogShipper::new(gate.clone(), transport.clone(), limits).unwrap()); let mut feed = shipper.take_publication_feed().unwrap(); - for index in 0..MAX_QUEUED_SUBMISSIONS as u64 { + for index in 0..accepted { shipper .submit(publication_submission(&cuts, index)) .await .unwrap(); } - let next = publication_submission(&cuts, 512); - let running = Arc::clone(&shipper); - let blocked = tokio::spawn(async move { running.submit(next).await }); - // The ordered lane is held only after native loading, while the full - // publication queue refuses a slot. Observe that actual blocked state. - tokio::time::timeout(Duration::from_secs(1), async { - while shipper.order.try_lock().is_ok() { - tokio::task::yield_now().await; - } - }) - .await - .unwrap(); + let next = publication_submission(&cuts, accepted); + let blocked = shipper.submit(next); + tokio::pin!(blocked); + // Poll the actual admission waiter. A full byte window cannot hold the + // global ordering lock or issue a ticket while this future is pending. + assert!(futures_util::poll!(blocked.as_mut()).is_pending()); + assert!(shipper.order.try_lock().is_ok()); tokio::time::timeout(Duration::from_secs(1), shipper.shutdown()) .await .unwrap() .unwrap(); - assert!(matches!(blocked.await.unwrap(), Err(Error::RuntimeClosed))); - assert_eq!(gate.issued_through(), 512); + assert!(matches!(blocked.await, Err(Error::RuntimeClosed))); + assert_eq!(gate.issued_through(), accepted); let mut captures = 0; while let Some(capture) = feed.recv().await { captures += 1; capture.assignment().verify(capture.frames()).unwrap(); } - assert_eq!(captures, 512); - assert_eq!(transport.batch_sizes(node(2)).iter().sum::(), 512); + assert_eq!(captures, accepted); + assert_eq!( + transport.batch_sizes(node(2)).iter().sum::() as u64, + accepted + ); } #[tokio::test] @@ -577,9 +654,14 @@ async fn covered_queued_prefix_keeps_the_uncovered_suffix_fleet_durable() { }) .collect(); - append_batch(&gate, transport, leader, 2, &[member], batch) - .await - .unwrap(); + let lanes = replication::MemberLanes::start(transport, leader, 2, vec![member], 1 << 30); + let round = lanes.enqueue(batch, gate.tiered_through()).unwrap().await; + assert!(complete_round( + &gate, + &crate::fleet::telemetry::CellTelemetryHandle::default(), + round, + )); + lanes.join().await.unwrap(); assert_eq!( gate.prove(second).await.unwrap().source(), @@ -775,3 +857,135 @@ async fn encoding_failure_does_not_consume_a_node_sequence() { ); shipper.shutdown().await.unwrap(); } + +#[tokio::test] +async fn ordered_member_lanes_advance_independently_and_group_queued_rounds() { + let (_directory, cuts) = capture(); + let gate = DurabilityGate::new(session(1), node(1), 2, [node(2), node(3)]).unwrap(); + gate.activate_fleet().unwrap(); + let entered = Arc::new(tokio::sync::Notify::new()); + let release = Arc::new(tokio::sync::Notify::new()); + let transport = Arc::new(RecordingTransport { + held: Some((node(2), entered.clone(), release.clone())), + ..RecordingTransport::default() + }); + let shipper = NodeLogShipper::new( + gate.clone(), + transport.clone(), + cellule_ltx::Limits::default(), + ) + .unwrap(); + let first = shipper + .submit(publication_submission(&cuts, 0)) + .await + .unwrap(); + tokio::time::timeout(Duration::from_secs(1), entered.notified()) + .await + .unwrap(); + let mut tickets = vec![first]; + let mut independent = true; + for index in 1..=5 { + let ticket = shipper + .submit(publication_submission(&cuts, index)) + .await + .unwrap(); + tickets.push(ticket); + let progress = tokio::time::timeout(Duration::from_millis(100), async { + loop { + let reached = + transport + .batches + .lock() + .unwrap() + .iter() + .any(|(member, sequences)| { + *member == node(3) + && sequences + .last() + .is_some_and(|last| *last >= ticket.last_sequence()) + }); + if reached { + break; + } + tokio::task::yield_now().await; + } + }) + .await; + if progress.is_err() { + independent = false; + break; + } + } + // A fast member alone grants no Fleet proof past the held original round. + let unproven = tokio::time::timeout(Duration::from_millis(10), gate.wait_followers(first)) + .await + .is_err(); + release.notify_one(); + for ticket in tickets { + gate.wait_followers(ticket).await.unwrap(); + } + shipper.shutdown().await.unwrap(); + assert!( + independent, + "a slow member must not prevent the other ordered lane from accepting later rounds" + ); + assert!(unproven); + assert_eq!(transport.batch_sizes(node(2)), [1, 5]); + assert_eq!(transport.batch_sizes(node(3)), [1, 1, 1, 1, 1, 1]); +} + +#[tokio::test] +async fn cancelled_shutdown_waiter_does_not_make_a_later_join_complete_early() { + let (_directory, cuts) = capture(); + let gate = DurabilityGate::new(session(1), node(1), 2, [node(2)]).unwrap(); + let entered = Arc::new(tokio::sync::Notify::new()); + let release = Arc::new(tokio::sync::Notify::new()); + let transport = Arc::new(RecordingTransport { + held: Some((node(2), entered.clone(), release.clone())), + ..RecordingTransport::default() + }); + let shipper = + Arc::new(NodeLogShipper::new(gate, transport, cellule_ltx::Limits::default()).unwrap()); + shipper.submit(submission(&cuts)).await.unwrap(); + tokio::time::timeout(Duration::from_secs(1), entered.notified()) + .await + .unwrap(); + let closing = shipper.clone(); + let waiter = tokio::spawn(async move { closing.shutdown().await }); + tokio::time::timeout(Duration::from_secs(1), async { + while shipper.worker.lock().unwrap().is_some() { + tokio::task::yield_now().await; + } + }) + .await + .unwrap(); + waiter.abort(); + assert!(waiter.await.unwrap_err().is_cancelled()); + let next = shipper.shutdown(); + tokio::pin!(next); + let premature = match futures_util::poll!(next.as_mut()) { + std::task::Poll::Ready(result) => { + result.unwrap(); + true + } + std::task::Poll::Pending => false, + }; + release.notify_one(); + if !premature { + tokio::time::timeout(Duration::from_secs(1), next) + .await + .unwrap() + .unwrap(); + } + tokio::time::timeout(Duration::from_secs(1), async { + while shipper.bytes.available_permits() != shipper.max_outstanding_bytes as usize { + tokio::task::yield_now().await; + } + }) + .await + .unwrap(); + assert!( + !premature, + "a cancelled join waiter cannot manufacture a later successful shutdown before accepted member I/O finishes" + ); +} From a42d344ae44cadd93c320724acb3e9a149ba9935 Mon Sep 17 00:00:00 2001 From: forhappy Date: Fri, 9 Oct 2026 11:04:01 -0700 Subject: [PATCH 082/102] Reuse freshly matched bundle metadata during selection --- .../docs/write-performance-design.md | 6 ++- .../cellule-runtime/src/node/bundle/origin.rs | 31 ++++++++++++++ .../src/node/bundle/selection.rs | 42 ++----------------- .../node/bundle/tests/index/history_cohort.rs | 27 ++++++++++++ docs/pr67-publication-path-diagnosis.md | 2 +- 5 files changed, 68 insertions(+), 40 deletions(-) diff --git a/crates/cellule-runtime/docs/write-performance-design.md b/crates/cellule-runtime/docs/write-performance-design.md index a1b16827..9c24355d 100644 --- a/crates/cellule-runtime/docs/write-performance-design.md +++ b/crates/cellule-runtime/docs/write-performance-design.md @@ -79,7 +79,11 @@ and accepted member I/O keep that credit through their original lifetimes. Eight original rounds can overlap in per-member FIFO lanes, with ordered credit application and bounded grouping into canonical follower append/fsync. Every lane joins before epoch shutdown, including after cancellation of a join waiter. -This changes native execution; the bundle publisher still runs expensive +Selection reuses its encoder-checked binding metadata only after this operation's +complete fresh origin read matches the proposal bytes. It avoids a second catalog +decode and hydrates no duplicate selected histories; referenced base and native +history bodies still pass fresh canonical verification before CAS. This changes +native execution and selection CPU work; the bundle publisher still runs expensive selection cohorts serially. Publication staging, reduced repeated verification and complete application performance qualification remain required. Component passes establish no new TPS claim. diff --git a/crates/cellule-runtime/src/node/bundle/origin.rs b/crates/cellule-runtime/src/node/bundle/origin.rs index 3bae012d..5908ce6b 100644 --- a/crates/cellule-runtime/src/node/bundle/origin.rs +++ b/crates/cellule-runtime/src/node/bundle/origin.rs @@ -38,6 +38,37 @@ impl OriginBundle { }) } + pub(super) fn selected_bindings(&self, prepared: &PreparedNodeBundle) -> Result> { + if self.session != prepared.catalog.session || self.head != prepared.head { + return Err(Error::Node("bundle metadata origin differs")); + } + // Encoding checked these exact private rows and their canonical extents. + // This operation's complete fresh read matched the encoded bytes before + // this view can exist. Re-decoding their shards/histories repeats that + // work; it supplies no additional origin observation. Dependency bodies + // remain independently verified in the canonical cohort verifier. + Ok(prepared + .catalog + .bindings + .iter() + .filter(|binding| { + binding + .locators + .iter() + .any(|locator| locator.object.is_none()) + }) + .cloned() + .map(|mut binding| { + for locator in &mut binding.locators { + if locator.object.is_none() { + locator.object = Some(self.head.digest); + } + } + binding + }) + .collect()) + } + pub(super) fn range( &self, session: SessionId, diff --git a/crates/cellule-runtime/src/node/bundle/selection.rs b/crates/cellule-runtime/src/node/bundle/selection.rs index e273e44e..e2e6a03c 100644 --- a/crates/cellule-runtime/src/node/bundle/selection.rs +++ b/crates/cellule-runtime/src/node/bundle/selection.rs @@ -215,46 +215,12 @@ impl NodeDirectory { if prepared.assignments.is_empty() { return Err(Error::Node("native bundle has no complete assignments")); } - let cells = prepared - .catalog - .bindings - .iter() - .filter(|binding| { - binding - .locators - .iter() - .any(|locator| locator.object.is_none()) - }) - .map(|binding| { - ( - *binding.application.as_bytes(), - *binding.control.cell.as_bytes(), - ) - }) - .collect(); let origin = origin::OriginBundle::load(&self.layout, prepared).await?; - let catalog = index::load_cells( - &self.layout, - prepared.catalog.session, - prepared.head, - &cells, - Some(&origin), - ) - .await?; - let bindings = catalog - .bindings - .into_iter() - .filter(|binding| { - cells.contains(&( - *binding.application.as_bytes(), - *binding.control.cell.as_bytes(), - )) - }) - .collect::>(); + let bindings = origin.selected_bindings(prepared)?; verification::verify_cohort( &self.layout, - catalog.session, - catalog.epoch, + prepared.catalog.session, + prepared.catalog.epoch, &bindings, limits, &origin, @@ -270,7 +236,7 @@ impl NodeDirectory { pin, binding, head: prepared.head, - session: catalog.session, + session: prepared.catalog.session, live: None, }); } diff --git a/crates/cellule-runtime/src/node/bundle/tests/index/history_cohort.rs b/crates/cellule-runtime/src/node/bundle/tests/index/history_cohort.rs index 1ac706b2..e222d113 100644 --- a/crates/cellule-runtime/src/node/bundle/tests/index/history_cohort.rs +++ b/crates/cellule-runtime/src/node/bundle/tests/index/history_cohort.rs @@ -123,6 +123,24 @@ async fn selection_groups_fresh_historical_extents_across_sixty_four_cells() { .prepare_node_bundle(&f.node, &frames, &assignments, now) .await .unwrap(); + let wanted = cells + .iter() + .map(|cell| { + ( + *cell.authority.layout().application_id(), + *cell.control.value().cell.as_bytes(), + ) + }) + .collect(); + let decoded = catalog_index::load_cells( + &f.layout, + prepared.catalog.session, + prepared.head, + &wanted, + None, + ) + .await + .unwrap(); f.count.reset(); let (node, proofs) = f .directory @@ -148,6 +166,15 @@ async fn selection_groups_fresh_historical_extents_across_sixty_four_cells() { ); assert_eq!(proofs.len(), MAX_FRAMES); for proof in &proofs { + let canonical = decoded + .bindings + .iter() + .find(|binding| binding.control.bundle_binding == Some(proof.binding())) + .unwrap(); + assert_eq!( + &proof.binding, canonical, + "selected metadata must equal an independent fresh canonical decode" + ); assert_eq!(proof.commit_sequence(), 3); assert_eq!(proof.locator_count(), 2); let restored = proof diff --git a/docs/pr67-publication-path-diagnosis.md b/docs/pr67-publication-path-diagnosis.md index 4a402c23..1b771e35 100644 --- a/docs/pr67-publication-path-diagnosis.md +++ b/docs/pr67-publication-path-diagnosis.md @@ -38,7 +38,7 @@ append batch, and its example HTTP adapter holds a member grant mutex across the RPC. Concurrent futures alone would not establish safe delivery ordering. Publication also does different work. Celld's -[bundle flush](https://github.com/denoland/celld/blob/f2bf648663a610eefde71f3547ad61e9b896b1f0/crates/celld/node_log.rs#L6436) +[bundle flush](https://github.com/denoland/celld/blob/f2bf648663a610eefde71f3547ad61e9b896b1f0/crates/celld/node_log.rs#L6628) uploads new native entries, then reads the original log record to check that the epoch remains open before crediting the upload. Cellule reloads selected catalog metadata, builds and uploads a proposal, freshly verifies its origin, From 3cad11e42f484c53b200f3205f5f53e0d63ef8a7 Mon Sep 17 00:00:00 2001 From: forhappy Date: Fri, 9 Oct 2026 11:51:05 -0700 Subject: [PATCH 083/102] Preserve a short group commit window in the native pipeline --- crates/cellule-runtime/docs/failover-and-followers.md | 8 +++++--- crates/cellule-runtime/src/node/log_shipper/mod.rs | 5 ++++- 2 files changed, 9 insertions(+), 4 deletions(-) diff --git a/crates/cellule-runtime/docs/failover-and-followers.md b/crates/cellule-runtime/docs/failover-and-followers.md index ce2585c9..3f26dd90 100644 --- a/crates/cellule-runtime/docs/failover-and-followers.md +++ b/crates/cellule-runtime/docs/failover-and-followers.md @@ -284,7 +284,9 @@ set, activation bit, contiguous object watermark, and renewable recovery claim. - Reserves capture bytes and bounded queue bookkeeping before assigning a sequence. - Multiplexes accepted cuts in submission order. -- Batches for at most one millisecond or 64 frames. +- Batches for at most four milliseconds or 64 frames. The short window preserves + group commit when the ordered pipeline can dispatch before a preceding RPC + completes; full or byte-limited batches dispatch immediately. - Enqueues each batch synchronously into every member's ordered FIFO, with at most eight original rounds in flight. Member I/O progresses independently. - Groups already queued adjacent requests into one canonical follower append, @@ -1925,8 +1927,8 @@ charged to the local disk ledger and physical backlog high water. At 1,000 aggregate transactions/s, the design must group fsync work: -- The leader batches frames across Cells for up to the smaller of one - millisecond, 64 frames, or 64 MiB. +- The leader batches frames across Cells for up to the smaller of four + milliseconds, 64 frames, or the native-byte limit. - Each follower appends that batch and performs one `sync_data`. - The exact interval is a measured runtime constant, not a per-deployment tuning surface until qualification proves one is needed. diff --git a/crates/cellule-runtime/src/node/log_shipper/mod.rs b/crates/cellule-runtime/src/node/log_shipper/mod.rs index e69c0acb..155b04a7 100644 --- a/crates/cellule-runtime/src/node/log_shipper/mod.rs +++ b/crates/cellule-runtime/src/node/log_shipper/mod.rs @@ -20,7 +20,10 @@ pub(crate) use publication::{SelectedBundle, SubmittedCapture}; const MAX_BATCH_FRAMES: usize = 64; const MAX_QUEUED_SUBMISSIONS: usize = 512; const NODE_FRAME_HEADER_BYTES: u64 = 240; -const BATCH_INTERVAL: Duration = Duration::from_millis(1); +// An ordered pipeline no longer accumulates submissions while waiting for a +// preceding RPC. Give small captures a bounded group-commit window instead of +// sending every newly available round as another tiny follower request. +const BATCH_INTERVAL: Duration = Duration::from_millis(4); type WorkerResult = std::result::Result<(), Arc>; /// One captured Cell commit awaiting ordered node-log assignment. From d5cec4885e97228efcf5c4013cd70ebc5eb79af8 Mon Sep 17 00:00:00 2001 From: forhappy Date: Fri, 9 Oct 2026 12:08:48 -0700 Subject: [PATCH 084/102] Credit completed follower rounds while assembling the next batch --- .../src/node/log_shipper/mod.rs | 37 +++++++++++--- .../src/node/log_shipper/tests.rs | 50 +++++++++++++++++++ .../runtime/lifecycle/durability/recovery.rs | 4 ++ 3 files changed, 84 insertions(+), 7 deletions(-) diff --git a/crates/cellule-runtime/src/node/log_shipper/mod.rs b/crates/cellule-runtime/src/node/log_shipper/mod.rs index 155b04a7..3ff1c544 100644 --- a/crates/cellule-runtime/src/node/log_shipper/mod.rs +++ b/crates/cellule-runtime/src/node/log_shipper/mod.rs @@ -2,7 +2,7 @@ use std::{collections::VecDeque, io::Read as _, sync::Arc, time::Duration}; use bytes::Bytes; -use futures_util::stream::StreamExt; +use futures_util::{FutureExt, stream::StreamExt}; use tokio::sync::{OwnedSemaphorePermit, Semaphore, mpsc}; use crate::identity::{ApplicationId, CellId}; @@ -567,6 +567,18 @@ async fn run_rounds( let mut closed = false; let mut rounds = futures_util::stream::FuturesOrdered::new(); loop { + // Member tasks can have completed while this dispatcher was receiving + // captures. Credit ready original rounds before assembling more work. + while !rounds.is_empty() { + let Some(Some(round)) = rounds.next().now_or_never() else { + break; + }; + if !complete_round(gate, telemetry, round) { + stop_shipper(gate, bytes, stopping); + receiver.close(); + return; + } + } if closed && pending.is_empty() && rounds.is_empty() { bytes.close(); stopping.send_replace(true); @@ -633,13 +645,24 @@ async fn run_rounds( { break; } - match tokio::time::timeout_at(deadline, receiver.recv()).await { - Ok(Some(submission)) => pending.extend(submission.frames), - Ok(None) => { - closed = true; - break; + tokio::select! { + round = rounds.next(), if !rounds.is_empty() => { + if let Some(round) = round + && !complete_round(gate, telemetry, round) + { + stop_shipper(gate, bytes, stopping); + receiver.close(); + return; + } + } + submission = tokio::time::timeout_at(deadline, receiver.recv()) => match submission { + Ok(Some(submission)) => pending.extend(submission.frames), + Ok(None) => { + closed = true; + break; + } + Err(_) => break, } - Err(_) => break, } } // Enqueue synchronously, in original sequence order, before returning a diff --git a/crates/cellule-runtime/src/node/log_shipper/tests.rs b/crates/cellule-runtime/src/node/log_shipper/tests.rs index 7d493e37..3e75025c 100644 --- a/crates/cellule-runtime/src/node/log_shipper/tests.rs +++ b/crates/cellule-runtime/src/node/log_shipper/tests.rs @@ -858,6 +858,56 @@ async fn encoding_failure_does_not_consume_a_node_sequence() { shipper.shutdown().await.unwrap(); } +#[tokio::test] +async fn completed_follower_round_is_credited_during_next_batch_assembly() { + let (_directory, cuts) = capture(); + let gate = DurabilityGate::new(session(1), node(1), 2, [node(2)]).unwrap(); + gate.activate_fleet().unwrap(); + let entered = Arc::new(tokio::sync::Notify::new()); + let release = Arc::new(tokio::sync::Notify::new()); + let transport = Arc::new(RecordingTransport { + held: Some((node(2), entered.clone(), release.clone())), + ..RecordingTransport::default() + }); + let shipper = NodeLogShipper::start( + gate.clone(), + transport, + cellule_ltx::Limits::default(), + crate::fleet::telemetry::CellTelemetryHandle::default(), + Duration::from_millis(250), + ) + .unwrap(); + let first = shipper + .submit(publication_submission(&cuts, 0)) + .await + .unwrap(); + entered.notified().await; + let second = shipper + .submit(publication_submission(&cuts, 1)) + .await + .unwrap(); + let sender = shipper.sender.lock().unwrap().clone().unwrap(); + // The first member I/O is held. Reclaimed FIFO capacity establishes that + // the dispatcher has received the second capture and begun its assembly. + tokio::time::timeout(Duration::from_secs(1), async { + while sender.capacity() != MAX_QUEUED_SUBMISSIONS { + tokio::task::yield_now().await; + } + }) + .await + .unwrap(); + drop(sender); + release.notify_one(); + let credited = + tokio::time::timeout(Duration::from_millis(50), gate.wait_followers(first)).await; + gate.wait_followers(second).await.unwrap(); + shipper.shutdown().await.unwrap(); + assert!( + matches!(credited, Ok(Ok(_))), + "a completed original round must not wait for the next assembly deadline" + ); +} + #[tokio::test] async fn ordered_member_lanes_advance_independently_and_group_queued_rounds() { let (_directory, cuts) = capture(); diff --git a/crates/cellule-runtime/tests/runtime/lifecycle/durability/recovery.rs b/crates/cellule-runtime/tests/runtime/lifecycle/durability/recovery.rs index 5ae2bf3e..7a9b1ebe 100644 --- a/crates/cellule-runtime/tests/runtime/lifecycle/durability/recovery.rs +++ b/crates/cellule-runtime/tests/runtime/lifecycle/durability/recovery.rs @@ -113,6 +113,10 @@ async fn lost_ack_suffix_recovers_an_ambiguous_command_without_reexecution() { .unwrap(); if current.value().root.as_ref().unwrap().commit_sequence == 1 && runtime.stats().unpublished_node_log_bytes() == 0 + // An object response can precede the first follower append. + // Establish that append before submitting the lost-ACK cut; + // otherwise group commit can acknowledge both as its first RPC. + && lost_ack_transport.acknowledged_once.load(Ordering::Acquire) { break; } From 3ed5aafde2da636ee03d0368a4336a4b9ce53603 Mon Sep 17 00:00:00 2001 From: forhappy Date: Fri, 9 Oct 2026 12:54:42 -0700 Subject: [PATCH 085/102] Keep the crash child marker independent of the test harness line --- crates/cellule-ltx/tests/ltx/crash.rs | 4 +++- 1 file changed, 3 insertions(+), 1 deletion(-) diff --git a/crates/cellule-ltx/tests/ltx/crash.rs b/crates/cellule-ltx/tests/ltx/crash.rs index 8faf24f2..1f08a232 100644 --- a/crates/cellule-ltx/tests/ltx/crash.rs +++ b/crates/cellule-ltx/tests/ltx/crash.rs @@ -99,8 +99,10 @@ fn crash_writer() { .iter() .map(|byte| format!("{byte:02x}")) .collect(); + // With one test thread, libtest leaves its test name on this stdout line. + // Start the child protocol on a fresh line so the parent can recognize it. println!( - "LTX-CUT {} {} {} {} {} {} {} {}", + "\nLTX-CUT {} {} {} {} {} {} {} {}", info.min_txid, info.max_txid, info.page_size, From 6fa5278c7d0af401ed025365610a710c5daa0469 Mon Sep 17 00:00:00 2001 From: forhappy Date: Fri, 9 Oct 2026 12:54:55 -0700 Subject: [PATCH 086/102] Record native pipeline measurements and remaining qualification failures --- .../docs/write-performance-design.md | 22 ++- docs/pr67-native-pipeline-measurement.md | 165 ++++++++++++++++++ docs/pr67-publication-path-diagnosis.md | 19 +- 3 files changed, 194 insertions(+), 12 deletions(-) create mode 100644 docs/pr67-native-pipeline-measurement.md diff --git a/crates/cellule-runtime/docs/write-performance-design.md b/crates/cellule-runtime/docs/write-performance-design.md index 9c24355d..51246c5a 100644 --- a/crates/cellule-runtime/docs/write-performance-design.md +++ b/crates/cellule-runtime/docs/write-performance-design.md @@ -1,6 +1,19 @@ # Node write and read performance design -The [latest fresh write comparison](../../../docs/pr67-base-pipeline-measurement.md) +The [native-pipeline comparison](../../../docs/pr67-native-pipeline-measurement.md) +completes 468.38 Fleet writes/s for unchanged production, 404.68/s for the first +pipeline candidate, 553.53/s for the corrected candidate and 1,993.08/s for celld. +The correction preserves a four-millisecond assembly window and credits ready +follower rounds during the next assembly. Compared with the baseline reference, +its observed TPS is 18.18% higher and successful scheduled p99 falls +574.13→321.43 ms; one short later run establishes no repeatable causal gain. +Ordered-lock wait falls 120.61→0.00045 ms. All Cellule warm ACK audits fail with +HTTP 503s; none reaches cold audit. The corrected owner exceeds the original +120-second process drain limit. Celld passes all 181,598 warm/cold mutations and +original retries but drops 403 offers. These diagnostics establish no qualified +capacity. Publication cost, complete draining and availability remain unresolved. + +The [preceding fresh write comparison](../../../docs/pr67-base-pipeline-measurement.md) completes 579.10 Fleet writes/s for unchanged production code, 570.25/s for a continuous-base-slot trial and 1,999.83/s for celld. The trial has no acceptable TPS/latency gain and is reverted. Both Cellule cases fail warm ACK availability @@ -22,9 +35,10 @@ a draft. The [preceding encoder comparison](../../../docs/pr67-encoder-cost-meas and its passing and failed evidence remain separate observations. The [submission diagnosis](../../../docs/pr67-submission-timing-measurement.md) -identified publication capacity held under the global issuance lock. Observed ordered wait remains about 119 ms; the latest phase counts -straddle window boundaries and cannot support an exact successful-submission -partition. This coupling queues native progress before follower proof starts. Repeated +identified publication capacity held under the global issuance lock. Earlier +phase counts straddled window boundaries and could not support an exact +successful-submission partition. The new byte-bounded feed removes that +slot wait from global ordering, as measured above. Repeated historical/base verification and sparse root checkpoints keep the publication consumer expensive. Matching celld requires reducing that work and separating native progress from bounded recoverable publication debt; larger queues alone diff --git a/docs/pr67-native-pipeline-measurement.md b/docs/pr67-native-pipeline-measurement.md new file mode 100644 index 00000000..3a130610 --- /dev/null +++ b/docs/pr67-native-pipeline-measurement.md @@ -0,0 +1,165 @@ +# PR 67: native admission and ordered follower pipeline + +**Performance parity remains unmet; PR #67 stays a draft.** Native admission no +longer waits for a publication slot under the global ordering lock. The first +pipeline candidate regresses throughput. Preserving a short commit window and +crediting ready rounds during the next batch assembly restores throughput in a +later diagnostic. Availability, publication cost and complete draining still +fail qualification. + +## Implemented behavior + +- The original native byte window owns publication FIFO entries, capture control + allocations and member frame vectors. Capacity waits and local verification + precede ordered issuance; synchronous enqueue preserves the assigned range. +- Eight original rounds progress through independent ordered member lanes. + Adjacent delivered work uses the canonical follower append and group commit. + Credits stay ordered, require every member's exact coverage, and cannot jump + to a grouped receipt's later frontier. Accepted member I/O retains credit and + shutdown joins the original tasks even after a joining waiter is cancelled. +- The four-millisecond assembly window preserves batching. Ready follower rounds + receive ordered credit during the next assembly, without waiting for its + deadline. Frames, rounds and byte limits remain unchanged. +- Selection freshly reads and matches the complete uploaded bundle, then reuses + its private encoder-checked binding metadata within that operation. Fresh + base/root/history verification still precedes the canonical CAS; there is no + availability cache between selections. + +Celld provides the comparison for [ordered shipping](https://github.com/denoland/celld/blob/f2bf648663a610eefde71f3547ad61e9b896b1f0/crates/celld/ltx_repl.rs#L4530), +[follower group commit](https://github.com/denoland/celld/blob/f2bf648663a610eefde71f3547ad61e9b896b1f0/crates/celld/node_log.rs#L179) +and [new-entry bundle upload](https://github.com/denoland/celld/blob/f2bf648663a610eefde71f3547ad61e9b896b1f0/crates/celld/node_log.rs#L6628). +Its shipping concurrency is 64; this candidate retains eight rounds and one +ordered RPC at a time per member. The publication protocols remain different. + +Baseline `10370d20f52c0b2c6103b0df61c56a1252238d33` has unchanged production +`4a5b001` bytes. First candidate `a42d344ae44cadd93c320724acb3e9a149ba9935` +contains native admission, the pipeline and metadata reuse. Corrected candidate +`d5cec4885e97228efcf5c4013cd70ebc5eb79af8` adds the assembly window and prompt +round credit. The intervening `3cad11e` has no completed TPS measurement. + +## Matched write diagnostics + +Each case has 2,000 uniform Cells, 96-byte SQL values, the same request/result +ledger, 128 clients/queue slots, one owner and two followers, WAL NORMAL and +tmpfs. It offers 2,000 writes/s, with 30-second warmup and a 60-second measured +window. The pinned builder establishes distinct serving binaries and identical +driver, auditor, fixture bytes, images and runner. No build, contributor suite +or independent replay overlaps a timed window. + +Baseline, first candidate and celld run in the initial comparison. The corrected +candidate runs later against the same profile; its replay includes those +original baseline and celld journals. They are reference runs, not newly repeated +baseline/celld runs. These observations establish no repeatable causal gain. + +| Arm | Successful writes/s | Successful scheduled p99 ms | Errors | Dropped offers | +| --- | ---: | ---: | ---: | ---: | +| Cellule baseline | 468.38 | 574.13 | 52,505 | 39,266 | +| Cellule first pipeline candidate | 404.68 | 302.98 | 70,538 | 25,181 | +| Cellule corrected candidate | 553.53 | 321.43 | 60,319 | 26,404 | +| celld pinned reference | 1,993.08 | 67.96 | 0 | 403 | + +TPS counts only successful completions inside the measured window. Successful +scheduled p99 includes measured successful offers through client drain and +excludes fast failures. Successful request p99 is 381.28, 174.22, 173.65 and +48.91 ms respectively. The first candidate loses 13.60% TPS. The corrected +candidate completes 18.18% more TPS and has 44.01% lower scheduled p99 than the +baseline, but 6.09% higher p99 than the first candidate. The same baseline +binary's earlier 579.10/s result illustrates run variability. + +Independent replay reconciles every offer, attempt, original successful output, +payload byte count, per-Cell count, trailing completion and complete ACK +provenance. All 2,000 Cells have measured successes. The corrected candidate has +33,212 in-window successes, 65 trailing successes and a complete cohort of +61,233 ACKs including setup and warmup. + +| Arm | Complete ACK cohort | Warm audit | Cold audit / joined drain | +| --- | ---: | --- | --- | +| Baseline | 47,855 | 2,804 HTTP 503s | Not reached | +| First candidate | 47,222 | 17 HTTP 503s | Not reached | +| Corrected candidate | 61,233 | 32 HTTP 503s; 61,201 retries checked | Not reached; owner exit exceeds 120 s | +| celld | 181,598 | All mutations and original retries pass | All pass; drain 16.85 s | + +The corrected candidate's first recorded errors refer to Cell 243; the owner logs +`PendingPublication` during drain and the original process wait times out. +HTTP 503s fail availability qualification and do not establish lost data. +Celld's dropped offers and scheduled p99 also fail this run's unchanged +performance gates. No case qualifies. + +## What telemetry establishes + +Completed submission phase cohorts are 28,167, 24,303 and 33,277 respectively. +Each cohort's seven phase totals reconcile exactly with native submission time. + +| Mean window observation | Baseline | First candidate | Corrected candidate | +| --- | ---: | ---: | ---: | +| Ordered-lock wait, ms | 120.613 | 0.0011 | 0.00045 | +| Publication-slot wait, ms | 1.530 | 0.00038 | 0.00029 | +| Complete native submission, ms | 122.313 | 0.214 | 0.141 | +| Follower frames per sync | 27.64 / 27.64 | 6.25 / 6.22 | 11.39 / 11.40 | +| Follower sync calls | 1,088 / 1,088 | 4,161 / 4,177 | 3,090 / 3,088 | +| Follower-proof wait, ms, separate cohort | 23.84 | 52.37 | 51.93 | + +Caller waits overlap and are not serial TPS service intervals. Pipelining +removed incidental batching behind a preceding RPC. The short assembly window +increases group size relative to the first candidate, but RPC/sync work per +frame remains higher than the baseline. The two corrective changes are measured +together; this run cannot attribute their separate effects. + +Corrected-candidate publication debt starts/ends at 4,439/4,402 pending entries, +48.43/45.07 seconds oldest age and 6.23/9.50 MB retained captures. Unpublished +native-log bytes grow 32.36→48.05 MB. Active Cells fall 2,000→1,999. Boundary +states and a short window do not prove stable sustained debt. The measured +window has 113,890 immutable GETs and 103,097 node-authority range starts. +Selection still verifies previous roots/history and serializes with checkpoint +work. Metadata reuse removes a repeat decode; it does not implement celld's +new-entry-only publication protocol. + +## Verification and limits + +The held-publication 513-capture, independently progressing member and cancelled +shutdown-join regressions fail on unchanged production and pass on the first +candidate. Its 17 native shipping and 91 bundle tests pass; all 13 contributor +routes pass with 1,988 workspace tests/doctests, 60 local LTX tests and both Rust +1.97/1.99 Clippy checks. The 38 environment-dependent tests remain ignored. + +The ready-credit regression fails three times on `3cad11e` and passes three +times on `d5cec48`; all 18 native shipping tests pass. The lost-ACK fixture now +waits for its first successful follower append before issuing the command whose +ACK is lost. Previously nine of ten repetitions grouped both commands into the +successful first append and never injected the intended fault. The corrected +fixture passes ten repetitions with its original recovery assertions. + +The corrected candidate's first full suite fails a two-second Fleet-ACK wait +while root CAS is held. Ten isolated repetitions pass with an actual follower +receipt and zero selected publication, supporting a load-dependent timing +hypothesis without proving its cause. The one-thread controlled rerun exposes a +separate child protocol bug: libtest prefixes `LTX-CUT` on its test-name line, +so the parent misses the marker and times out. A fresh-line marker fixes the +fixture without changing its kill, restore assertions or 15-second deadline. +The corrected marker reproduces the original timeout and passes three fixed +repetitions. Final controlled verification uses one test thread and private +workstation scratch; all 13 contributor routes pass, with 1,989 passing +workspace test/doctest executions, 38 ignored environment tests, 60 local LTX +tests and both Clippy versions. The serving code is unchanged from measured +`d5cec48`; the later change is test-only. All failed runs remain part of the +evidence; no qualification expectation or original deadline changes. + +The shared VM has eight CPUs and 8,306,286,592 bytes of total memory across all +roles, with oversubscribed container ceilings. It does not qualify a dedicated +8-vCPU/16-GiB serving node. Cellule's original 64-MiB retained, 1-GiB disk and +20-MiB publication-working policies are unchanged; celld has no equivalent +fixture admission settings. No new Bucket, read-only, mixed or physical-media +results are supplied here. The three five-minute repetitions remain required. + +Full source/build identities, journals, audits, telemetry, valid and failed +attempts are outside Git under `/Volumes/Workspace/crabbuild-target/`: + +- `native-pipeline-20261009`: frozen initial comparison and contributor evidence; + 10,814 files, 1,041,121,919 bytes verified against its SHA-256 index. +- `native-batch-window-20261009`: frozen intervening failed suite and lost-ACK + injection diagnosis; 3,925 files, 54,464,448 bytes verified. +- `native-round-credit-20261009`: corrective regressions, pinned diagnostic, + independent journal replay, telemetry and contributor attempts. + +Final source/build hashes and qualification status are in each evidence index. +No raw journals or bulk artifacts are added to the repository. diff --git a/docs/pr67-publication-path-diagnosis.md b/docs/pr67-publication-path-diagnosis.md index 1b771e35..ba344b2b 100644 --- a/docs/pr67-publication-path-diagnosis.md +++ b/docs/pr67-publication-path-diagnosis.md @@ -2,16 +2,19 @@ **The fast paths are not architecturally equivalent yet.** Both systems use SQLite WAL, LTX captures, fenced ownership, follower logs and bucket storage. -Cellule still couples native admission to a serial, expensive publication -consumer. That consumer sets the write rate under sustained pressure. +The diagnosed baseline couples native admission to a serial, expensive +publication consumer. The [subsequent native-pipeline changes](pr67-native-pipeline-measurement.md) +remove its slot wait from global ordering and add independent ordered member +lanes. Their first measured candidate improves latency but regresses TPS; +serial publication and recoverable debt still limit sustained performance. Performance parity remains unmet and PR #67 stays a draft. ## What the code does -Cellule's [assignment path](../crates/cellule-runtime/src/node/log_shipper/mod.rs) +The baseline's [assignment path](https://github.com/crabbuild/cellule/blob/10370d20f52c0b2c6103b0df61c56a1252238d33/crates/cellule-runtime/src/node/log_shipper/mod.rs) acquires the global ordered lock, then awaits publication queue capacity before committing a sequence and enqueueing follower work. The -[publication feed](../crates/cellule-runtime/src/node/log_shipper/publication/mod.rs) +[publication feed](https://github.com/crabbuild/cellule/blob/10370d20f52c0b2c6103b0df61c56a1252238d33/crates/cellule-runtime/src/node/log_shipper/publication/mod.rs) has 512 submission slots and retains native-byte admission through selection. The consumer selects one cohort of at most 64 captures/frames and 4 MiB before starting the next. A full publication queue therefore blocks otherwise @@ -33,7 +36,7 @@ Moving the wait or enlarging the queue alone cannot raise sustainable service. Celld's [shipping loop](https://github.com/denoland/celld/blob/f2bf648663a610eefde71f3547ad61e9b896b1f0/crates/celld/ltx_repl.rs#L4530) keeps ordered rounds in flight independently of bucket work. Its [follower stream](https://github.com/denoland/celld/blob/f2bf648663a610eefde71f3547ad61e9b896b1f0/crates/celld/node_log.rs#L179) -groups already delivered requests into one durable append. Cellule awaits each +groups already delivered requests into one durable append. Baseline Cellule awaits each append batch, and its example HTTP adapter holds a member grant mutex across the RPC. Concurrent futures alone would not establish safe delivery ordering. @@ -45,10 +48,10 @@ catalog metadata, builds and uploads a proposal, freshly verifies its origin, base roots and historical chains, and selects a new head by authority CAS. These are different publication protocols and service costs. -## Cross-check against the latest retained run +## Cross-check against the preceding retained run Re-analysis of the unchanged `10370d2` run in the -[latest matched comparison](pr67-base-pipeline-measurement.md) verifies the +[preceding matched comparison](pr67-base-pipeline-measurement.md) verifies the start/end telemetry files against the frozen evidence index. This is an analysis of that existing 60-second run, not a new benchmark or optimization. @@ -146,7 +149,7 @@ The [older 144.50 versus 4,622.43 result](pr67-bounded-history-measurement.md) used 1,000 Cells and 15,000 offered writes/s. Celld dropped 622,398 offers and did not complete cold/drain qualification after an OOM during shutdown. It establishes neither sustainable 4,622-write/s capacity nor a comparison -with the current 2,000-Cell profile. The latest +with the 2,000-Cell profile. The preceding [uninstrumented production pair](pr67-base-pipeline-measurement.md) records 579.10 versus 1,999.83 writes/s, with Cellule errors, drops and failed warm ACK availability. The offered-load cap prevents either 2K reference from proving From f858ed3cdab6699cd9c5e837ba7025c9b36f80b4 Mon Sep 17 00:00:00 2001 From: forhappy Date: Fri, 9 Oct 2026 13:05:03 -0700 Subject: [PATCH 087/102] Expose original fence causes in the SQL diagnostic service --- Cargo.lock | 1 + crates/cellule-axum/Cargo.toml | 1 + crates/cellule-axum/README.md | 4 ++++ crates/cellule-axum/examples/sql.rs | 8 ++++++++ crates/cellule-runtime/src/cell/actor/requests.rs | 6 ++++++ 5 files changed, 20 insertions(+) diff --git a/Cargo.lock b/Cargo.lock index 716d0eb1..b4c0a7fa 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -265,6 +265,7 @@ dependencies = [ "tokio", "tokio-util", "tower", + "tracing-subscriber", "utoipa", "utoipa-axum", "uuid", diff --git a/crates/cellule-axum/Cargo.toml b/crates/cellule-axum/Cargo.toml index 25b3dd5e..b59a37e4 100644 --- a/crates/cellule-axum/Cargo.toml +++ b/crates/cellule-axum/Cargo.toml @@ -38,6 +38,7 @@ tempfile.workspace = true tokio = { workspace = true, features = ["macros", "net", "rt-multi-thread", "signal"] } tokio-util = { workspace = true, features = ["rt"] } tower = { version = "0.5", features = ["util"] } +tracing-subscriber = "0.3" [[example]] name = "typed-api-service" diff --git a/crates/cellule-axum/README.md b/crates/cellule-axum/README.md index fb3262af..1d5315b9 100644 --- a/crates/cellule-axum/README.md +++ b/crates/cellule-axum/README.md @@ -120,6 +120,10 @@ For REST routes `POST /orders` and `GET /orders/{id}`, run the cargo run -p cellule-axum --example sql --locked ``` +The SQL service writes runtime warnings, including the original cause when a +command fences its Cell, to stderr. Logging stays in the embedding application; +HTTP clients receive the same outcome-aware error envelope. + ## Write a manual handler Add the adapter alongside Axum and your Cellule application crates: diff --git a/crates/cellule-axum/examples/sql.rs b/crates/cellule-axum/examples/sql.rs index 0420a751..f18aa0f3 100644 --- a/crates/cellule-axum/examples/sql.rs +++ b/crates/cellule-axum/examples/sql.rs @@ -340,6 +340,14 @@ async fn example_storage() -> ExampleResult<(Store, Path)> { #[tokio::main] async fn main() -> ExampleResult<()> { + // The embedding application owns logging. Keep runtime fence causes on + // stderr while ordinary capacity refusals retain their HTTP error contract. + tracing_subscriber::fmt() + .with_max_level(tracing_subscriber::filter::LevelFilter::WARN) + .with_writer(std::io::stderr) + .with_ansi(false) + .try_init() + .map_err(std::io::Error::other)?; let fleet_config = fleet::Config::from_env()?; let cells = example_count("CELLULE_AXUM_CELLS", 1, MAX_CELLS)?; let workers = example_count("CELLULE_AXUM_WORKERS", 1, 16)?; diff --git a/crates/cellule-runtime/src/cell/actor/requests.rs b/crates/cellule-runtime/src/cell/actor/requests.rs index 4b20e17a..9b687717 100644 --- a/crates/cellule-runtime/src/cell/actor/requests.rs +++ b/crates/cellule-runtime/src/cell/actor/requests.rs @@ -369,6 +369,12 @@ pub(super) async fn prove_command( }; let fenced = result.is_err(); if fenced { + tracing::warn!( + cell = ?command.cell, + commit_sequence, + error = ?result.as_ref().err(), + "Cell durable confirmation fenced its owner" + ); let _ = pool.fence(command.cell).await; } let result = if fenced { From cf4785c675698afe786f566242d6d3dace32e775 Mon Sep 17 00:00:00 2001 From: forhappy Date: Fri, 9 Oct 2026 13:31:11 -0700 Subject: [PATCH 088/102] Reuse completed command admission for publication memory --- crates/cellule-runtime/docs/runtime.md | 9 +- .../cellule-runtime/src/cell/actor/group.rs | 42 ++++- .../cellule-runtime/src/cell/actor/handle.rs | 88 +++++++++ .../src/cell/actor/requests.rs | 22 ++- crates/cellule-runtime/src/fleet/resource.rs | 6 +- .../tests/runtime/lifecycle/durability.rs | 1 + .../runtime/lifecycle/durability/admission.rs | 8 +- .../lifecycle/durability/admission_handoff.rs | 172 ++++++++++++++++++ 8 files changed, 337 insertions(+), 11 deletions(-) create mode 100644 crates/cellule-runtime/tests/runtime/lifecycle/durability/admission_handoff.rs diff --git a/crates/cellule-runtime/docs/runtime.md b/crates/cellule-runtime/docs/runtime.md index 5c027718..cb94fada 100644 --- a/crates/cellule-runtime/docs/runtime.md +++ b/crates/cellule-runtime/docs/runtime.md @@ -65,7 +65,14 @@ sequence lane. That lane assigns consecutive envelope sequences and enqueues them atomically; failed validation consumes no ticket. Signing follows sequence assignment, and followers still verify the complete signed frames before fsync. -Each member retains its ordinary request and byte admission, and the group +Each member retains its ordinary request and byte admission through SQL. +Once the original worker returns, unused result allowance can transfer to the +exact publication cut, while the input and actual reply remain charged until +the command completes. Remaining unused node memory is released. A larger cut +still needs full fresh admission under the unchanged node limit; this handoff +does not grant durability or omit publication debt. Cancelled or timed-out +waiters retain the full allowance until their dispatched worker exits. +The group uses the existing database, capture, and retained-publication ceilings. A command error rolls back that member's savepoint. A whole transaction abort, capture failure, or lost publication proof never releases a new success; diff --git a/crates/cellule-runtime/src/cell/actor/group.rs b/crates/cellule-runtime/src/cell/actor/group.rs index dcd764b6..4462fc6a 100644 --- a/crates/cellule-runtime/src/cell/actor/group.rs +++ b/crates/cellule-runtime/src/cell/actor/group.rs @@ -98,15 +98,43 @@ pub(super) async fn execute( true, ) } else { + // Each member owns its own input and actual reply. The head's + // result can differ from the highest commit in the shared cut; + // use its original outcome when transferring its allowance. + let head_result_bytes = outcomes + .first() + .and_then(|outcome| outcome.as_ref().ok()) + .map_or(command.max_result_bytes, |outcome| outcome.result().len()); + let member_admission = + command + .group + .as_mut() + .ok_or(Error::Fenced) + .and_then(|group| { + for (member, outcome) in + group.members.iter_mut().zip(outcomes.iter().skip(1)) + { + if let Ok(outcome) = outcome { + member._work.finish_sql(outcome.result().len(), 0)?; + } + } + Ok(()) + }); if let Some(group) = &mut command.group { group.execution = Some(GroupOutcomes { outcomes, base_sequence, }); } - match pending { - Some(pending) => { - let result = match reserve_pending_publication(&pool, &pending) { + match (member_admission, pending) { + (Err(error), _) => (Err(error), true), + (Ok(()), Some(pending)) => { + let result = match reserve_pending_publication( + &pool, + &pending, + &mut command._work, + head_result_bytes, + ) { Ok(retained_reservation) => { let first = base_sequence.checked_add(1).ok_or(Error::Fenced); match first { @@ -131,7 +159,13 @@ pub(super) async fn execute( // this logical range, including durable rejections. (result, true) } - None => (Ok(CommandTaskResult::GroupRecorded), false), + (Ok(()), None) => ( + command + ._work + .finish_sql(head_result_bytes, 0) + .map(|_| CommandTaskResult::GroupRecorded), + false, + ), } } } diff --git a/crates/cellule-runtime/src/cell/actor/handle.rs b/crates/cellule-runtime/src/cell/actor/handle.rs index 128d357e..f993a8b6 100644 --- a/crates/cellule-runtime/src/cell/actor/handle.rs +++ b/crates/cellule-runtime/src/cell/actor/handle.rs @@ -59,12 +59,39 @@ pub(crate) struct CommandWork { pub(super) struct WorkAdmission { pub(super) kind: AdmissionKind, + operation_bytes: usize, pub(super) _request: Option, pub(super) _cell_bytes: OwnedSemaphorePermit, pub(super) _node_bytes: ResourceReservation, } impl WorkAdmission { + pub(super) fn finish_sql( + &mut self, + result_bytes: usize, + publication_bytes: usize, + ) -> crate::Result> { + let retained = self + .operation_bytes + .checked_add(result_bytes) + .ok_or(Error::Capacity("completed command bytes"))?; + let unused = self + ._node_bytes + .retained_bytes() + .checked_sub(retained) + .ok_or(Error::Capacity("completed result exceeds admission"))?; + // The original SQL job has exited. Its unused result allowance can + // own this exact cut without racing a second global admission. Keep + // the input and actual reply charged until their command is dropped. + let publication = if publication_bytes != 0 && publication_bytes <= unused { + Some(self._node_bytes.split_retained(publication_bytes)?) + } else { + None + }; + self._node_bytes.shrink_retained(retained)?; + Ok(publication) + } + pub(super) fn release_request_slot(&mut self) { // A finished request can hand its slot to the reply's next invocation. // Retained completion data still owns both byte reservations until drop. @@ -599,6 +626,7 @@ impl CellHandle { } let admission = WorkAdmission { kind, + operation_bytes, _request: Some(try_one( self.admission.requests.clone(), "Cell mailbox requests", @@ -657,3 +685,63 @@ pub(super) fn admission_error(error: TryAcquireError, resource: &'static str) -> TryAcquireError::NoPermits => Error::Capacity(resource), } } + +#[cfg(test)] +mod tests { + use super::*; + use crate::fleet::resource::ResourceLedger; + + fn admitted() -> (ResourceLedger, WorkAdmission) { + let ledger = ResourceLedger::new(ResourceCost::zero().with_retained_bytes(32)); + let work = WorkAdmission { + kind: AdmissionKind::Command, + operation_bytes: 8, + _request: Some(try_one(Arc::new(Semaphore::new(1)), "test request").unwrap()), + _cell_bytes: try_many(Arc::new(Semaphore::new(32)), 32, "test bytes").unwrap(), + _node_bytes: ledger + .try_reserve(ResourceCost::zero().with_retained_bytes(32)) + .unwrap(), + }; + (ledger, work) + } + + #[test] + fn completed_sql_transfers_only_unused_credit_and_preserves_both_lifetimes() { + let (ledger, mut work) = admitted(); + let publication = work.finish_sql(4, 12).unwrap().unwrap(); + assert_eq!(work._node_bytes.retained_bytes(), 12); + assert_eq!(publication.retained_bytes(), 12); + assert_eq!(ledger.snapshot().unwrap().used.retained_bytes(), 24); + drop(work); + assert_eq!(ledger.snapshot().unwrap().used.retained_bytes(), 12); + drop(publication); + assert_eq!(ledger.snapshot().unwrap().used, ResourceCost::zero()); + } + + #[test] + fn an_oversized_cut_requires_fresh_capacity_without_undercharging_the_reply() { + let (ledger, mut work) = admitted(); + assert!(work.finish_sql(4, 24).unwrap().is_none()); + assert_eq!(work._node_bytes.retained_bytes(), 12); + assert_eq!(ledger.snapshot().unwrap().used.retained_bytes(), 12); + assert!( + ledger + .try_reserve(ResourceCost::zero().with_retained_bytes(24)) + .is_err() + ); + drop(work); + assert_eq!(ledger.snapshot().unwrap().used, ResourceCost::zero()); + } + + #[test] + fn invalid_result_size_cannot_release_or_transfer_original_admission() { + let (ledger, mut work) = admitted(); + assert!(matches!( + work.finish_sql(25, 1), + Err(Error::Capacity("completed result exceeds admission")) + )); + assert_eq!(ledger.snapshot().unwrap().used.retained_bytes(), 32); + drop(work); + assert_eq!(ledger.snapshot().unwrap().used, ResourceCost::zero()); + } +} diff --git a/crates/cellule-runtime/src/cell/actor/requests.rs b/crates/cellule-runtime/src/cell/actor/requests.rs index 9b687717..9bfa9984 100644 --- a/crates/cellule-runtime/src/cell/actor/requests.rs +++ b/crates/cellule-runtime/src/cell/actor/requests.rs @@ -251,9 +251,20 @@ pub(super) async fn execute_command( succeeded = execution.is_ok(), ); let (result, must_fence) = match execution { - Ok(WorkerExecution::Recorded(outcome)) => (Ok(CommandTaskResult::Recorded(outcome)), false), + Ok(WorkerExecution::Recorded(outcome)) => ( + command + ._work + .finish_sql(outcome.result().len(), 0) + .map(|_| CommandTaskResult::Recorded(outcome)), + false, + ), Ok(WorkerExecution::Pending(pending)) => { - let result = reserve_pending_publication(&pool, &pending); + let result = reserve_pending_publication( + &pool, + &pending, + &mut command._work, + pending.outcome().result().len(), + ); let result = match result { Ok(retained_reservation) => durability .submit(pending.outcome().commit_sequence(), pending.cuts()) @@ -299,12 +310,19 @@ pub(super) async fn execute_command( pub(super) fn reserve_pending_publication( pool: &SqlWorkerPool, pending: &PendingCommit, + work: &mut WorkAdmission, + result_bytes: usize, ) -> crate::Result { // The host already owns the LTX file's disk reservation. Retain RAM for // shared indexes and live outcome/descriptor copies, not the on-disk body. // Physical backlog counters and their per-Cell limits remain unchanged. let bytes = usize::try_from(pending.retained_memory_bytes()) .map_err(|_| Error::Capacity("pending publication bytes"))?; + if let Some(reservation) = work.finish_sql(result_bytes, bytes)? { + return Ok(reservation); + } + // Larger cuts still need their full charged admission. Returning unused + // result capacity first makes it available without changing the node limit. pool.resource_ledger() .try_reserve(ResourceCost::zero().with_retained_bytes(bytes)) .map_err(|error| match error { diff --git a/crates/cellule-runtime/src/fleet/resource.rs b/crates/cellule-runtime/src/fleet/resource.rs index eaf56a5d..517fe70c 100644 --- a/crates/cellule-runtime/src/fleet/resource.rs +++ b/crates/cellule-runtime/src/fleet/resource.rs @@ -491,6 +491,10 @@ pub(crate) struct ResourceReservation { } impl ResourceReservation { + pub(crate) const fn retained_bytes(&self) -> usize { + self.cost.retained_bytes + } + /// Transfers already admitted memory to an independently owned lifetime. /// Total ledger usage is unchanged; this cannot admit after publication. pub(crate) fn split_retained(&mut self, bytes: usize) -> Result { @@ -506,7 +510,7 @@ impl ResourceReservation { }) } - /// Returns only memory whose owned capture indexes have already been dropped. + /// Returns memory after its original work exits or owned indexes are dropped. pub(crate) fn shrink_retained(&mut self, bytes: usize) -> Result<()> { let released = self .cost diff --git a/crates/cellule-runtime/tests/runtime/lifecycle/durability.rs b/crates/cellule-runtime/tests/runtime/lifecycle/durability.rs index d23d38e7..513ebd01 100644 --- a/crates/cellule-runtime/tests/runtime/lifecycle/durability.rs +++ b/crates/cellule-runtime/tests/runtime/lifecycle/durability.rs @@ -5,6 +5,7 @@ use cellule_runtime::fleet::telemetry::{CellTelemetry, CommandResponseSource, Pu mod admission; mod admission_batching; +mod admission_handoff; mod group; mod proofs; mod recovery; diff --git a/crates/cellule-runtime/tests/runtime/lifecycle/durability/admission.rs b/crates/cellule-runtime/tests/runtime/lifecycle/durability/admission.rs index 57f09ac4..a1bde63e 100644 --- a/crates/cellule-runtime/tests/runtime/lifecycle/durability/admission.rs +++ b/crates/cellule-runtime/tests/runtime/lifecycle/durability/admission.rs @@ -370,10 +370,12 @@ async fn durable_group_releases_all_request_slots_before_its_first_reply() { let admitted = futures_util::poll!(&mut query); followups.push((query, admitted)); } - // All completion data remains charged while its terminal request slots - // become available; releasing slots must not discharge retained bytes. - assert!(runtime.stats().retained_bytes() >= 4 * 2_048); + // SQL has exited and returned each unused result allowance. Every group's + // input (1,024 bytes) and actual reply (one byte) must remain charged while + // terminal request slots become available; releasing slots cannot drop them. + let retained = runtime.stats().retained_bytes(); resume.send(()).unwrap(); + assert!(retained >= 4 * (1_024 + 1), "retained: {retained}"); let mut observed = Vec::new(); for (query, admitted) in followups { observed.push(match admitted { diff --git a/crates/cellule-runtime/tests/runtime/lifecycle/durability/admission_handoff.rs b/crates/cellule-runtime/tests/runtime/lifecycle/durability/admission_handoff.rs new file mode 100644 index 00000000..a7cd0c83 --- /dev/null +++ b/crates/cellule-runtime/tests/runtime/lifecycle/durability/admission_handoff.rs @@ -0,0 +1,172 @@ +//! Completed SQL hands unused result admission to its exact publication cut. + +use super::*; +use std::time::Duration; + +struct SqlCompletionGate { + entered: Mutex>>, + resume: Mutex>, + responses: Mutex>, +} + +impl CellTelemetry for SqlCompletionGate { + fn command_execution(&self, _queue: Duration, _worker: Duration, succeeded: bool) { + assert!(succeeded); + if let Some(entered) = self.entered.lock().unwrap().take() { + entered.send(()).unwrap(); + // Pause after the original worker has returned its physical cut, + // before the actor reserves publication memory. Only this test's + // callback blocks; production telemetry must remain nonblocking. + self.resume + .lock() + .unwrap() + .recv_timeout(Duration::from_secs(10)) + .unwrap(); + } + } + + fn command_response(&self, source: CommandResponseSource, _: Duration, _: Duration) { + self.responses.lock().unwrap().push(source); + } +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn admitted_result_capacity_funds_publication_after_sql_under_full_memory_pressure() { + let fixture = fixture_for(b"publication-admission-handoff"); + let session = SessionId::from_bytes([144; 16]); + let member = NodeId::from_bytes([145; 16]); + let dirty = Arc::new(tokio::sync::Semaphore::new(1)); + let runtime = CellRuntime::new_with_replica_host_requiring_node_lease( + SqlWorkerPool::new(1, 1).unwrap(), + 2 * 1024 * 1024, + session, + ReplicaHost::default().with_dirty_slots(dirty.clone()), + ) + .unwrap(); + let lease = NodeLeaseGuard::new(0, 60_000).unwrap(); + runtime.install_node_lease(lease.clone()).unwrap(); + let directory = tempfile::TempDir::new().unwrap(); + let follower = cellule_runtime::FollowerStore::open( + directory.path().to_owned(), + Limits::default(), + DiskBudget::new(1 << 30), + ) + .unwrap(); + let transport: Arc = Arc::new( + cellule_runtime::node::log_transport::LocalFollowerTransport::new(member, follower), + ); + let gate = DurabilityGate::new( + session, + NodeId::from_bytes(*session.as_bytes()), + 1, + [member], + ) + .unwrap(); + let shipper = NodeLogShipper::new(gate.clone(), transport.clone(), Limits::default()).unwrap(); + runtime + .install_node_durability( + fixture.target.application(), + Arc::new(NodeDurability::new( + gate, + shipper, + Arc::new(TestNodeAuthority::default()), + transport, + lease, + )), + ) + .unwrap(); + let handle = bootstrap_on(&runtime, &fixture, session).await; + let occupied = dirty.clone().acquire_owned().await.unwrap(); + let (entered, entered_rx) = tokio::sync::oneshot::channel(); + let (resume, resume_rx) = mpsc::channel(); + let observations = Arc::new(SqlCompletionGate { + entered: Mutex::new(Some(entered)), + resume: Mutex::new(resume_rx), + responses: Mutex::new(Vec::new()), + }); + runtime.install_telemetry(observations.clone()).unwrap(); + let identity = mutation_identity_window(146, 10, 10_000); + let digest = Digest::from_bytes([146; 32]); + let first_handle = handle.clone(); + let first = tokio::spawn(async move { + first_handle + .execute(identity, digest, 20, 1_024, 1 << 20, |tx| { + tx.execute("UPDATE counter SET value = value + 1", [])?; + Ok(HandlerOutcome::Success(b"admitted".to_vec())) + }) + .await + }); + tokio::time::timeout(Duration::from_secs(5), entered_rx) + .await + .unwrap() + .unwrap(); + let stats = runtime.stats(); + let pressure = runtime + .try_reserve_node_bytes(stats.retained_capacity_bytes() - stats.retained_bytes()) + .unwrap(); + assert_eq!( + runtime.stats().retained_bytes(), + stats.retained_capacity_bytes() + ); + resume.send(()).unwrap(); + let outcome = tokio::time::timeout(Duration::from_secs(5), first).await; + let progress = runtime.publication_progress().await; + // Release every injected resource before checking results, including on + // the old post-commit admission failure, so shutdown joins original work. + drop(pressure); + drop(occupied); + let query = handle + .query(16, 16, |db| { + let value: i64 = db.query_row("SELECT value FROM counter", [], |row| row.get(0))?; + Ok(value.to_be_bytes().to_vec()) + }) + .await; + let retry = handle + .execute(identity, digest, 21, 1_024, 1 << 20, |_| { + panic!("a durable retry must never execute the SQL callback") + }) + .await; + let shutdown = runtime.shutdown().await; + + let outcome = outcome.unwrap().unwrap().unwrap(); + assert_eq!(outcome.commit_sequence(), 1); + assert_eq!(outcome.result(), b"admitted"); + let progress = progress.unwrap(); + assert_eq!(progress.pending_publications, 1); + assert!(progress.retained_capture_bytes > 0); + assert_eq!(query.unwrap(), 1_i64.to_be_bytes()); + assert_eq!(retry.unwrap(), outcome); + assert_eq!( + observations.responses.lock().unwrap().as_slice(), + &[ + CommandResponseSource::Fleet, + CommandResponseSource::Recorded + ] + ); + shutdown.unwrap(); + assert_eq!(runtime.stats().retained_bytes(), 0); + assert_eq!(dirty.available_permits(), 1); + + let control = CellAuthority::new(fixture.layout.clone()) + .load(fixture.target.cell_id()) + .await + .unwrap() + .unwrap(); + assert_eq!(control.value().state, ControlState::Idle); + let root = control.value().ltx_root().unwrap(); + assert_eq!(root.commit_sequence, 1); + let restored = fixture._directory.path().join("handoff-restored.sqlite"); + let verified = fixture.replica.open_root(&root).await.unwrap(); + assert_eq!(verified.restore(&restored).await.unwrap(), root.position); + let mut recovered = cellule_runtime::cell::executor::CellExecutor::new( + cellule_ltx::Db::open(&restored, Limits::default()).unwrap(), + fixture.target.cell_id(), + IncarnationId::from_bytes([2; 16]), + 1, + ); + assert_eq!( + recovered.resolve(identity, digest, 22, 1 << 20).unwrap(), + Resolution::Committed(outcome) + ); + recovered.close().unwrap(); +} From 1f3d78d1b590eee185c24b7aaba126087b8b9126 Mon Sep 17 00:00:00 2001 From: forhappy Date: Fri, 9 Oct 2026 14:10:54 -0700 Subject: [PATCH 089/102] Record publication admission recovery and write regressions --- .../docs/write-performance-design.md | 9 ++ docs/pr67-native-pipeline-measurement.md | 4 + .../pr67-publication-admission-measurement.md | 149 ++++++++++++++++++ 3 files changed, 162 insertions(+) create mode 100644 docs/pr67-publication-admission-measurement.md diff --git a/crates/cellule-runtime/docs/write-performance-design.md b/crates/cellule-runtime/docs/write-performance-design.md index 51246c5a..30b9917d 100644 --- a/crates/cellule-runtime/docs/write-performance-design.md +++ b/crates/cellule-runtime/docs/write-performance-design.md @@ -1,5 +1,14 @@ # Node write and read performance design +The latest [publication-admission measurement](../../../docs/pr67-publication-admission-measurement.md) +identifies a post-SQL RAM reservation race and transfers unused original command +credit to publication. All 47,471 candidate ACKs pass warm/cold mutation and +original retry audits; fleet drain takes 47.68 seconds. Write performance +regresses in this short pair: 481.80→451.70 successful writes/s and +285.52→1,285.01-ms successful scheduled p99; fresh celld completes 1,996.33/s. +Publication debt and extensive root/history I/O remain. These are diagnostics, +not performance parity or qualification. The acceptance gates below are unchanged. + The [native-pipeline comparison](../../../docs/pr67-native-pipeline-measurement.md) completes 468.38 Fleet writes/s for unchanged production, 404.68/s for the first pipeline candidate, 553.53/s for the corrected candidate and 1,993.08/s for celld. diff --git a/docs/pr67-native-pipeline-measurement.md b/docs/pr67-native-pipeline-measurement.md index 3a130610..a0e546fc 100644 --- a/docs/pr67-native-pipeline-measurement.md +++ b/docs/pr67-native-pipeline-measurement.md @@ -1,5 +1,9 @@ # PR 67: native admission and ordered follower pipeline +The later [publication-admission measurement](pr67-publication-admission-measurement.md) +fixes a demonstrated owner-fencing race and passes its full ACK/drain/cold audit, +but records worse TPS and successful p99. Results below remain historical observations. + **Performance parity remains unmet; PR #67 stays a draft.** Native admission no longer waits for a publication slot under the global ordering lock. The first pipeline candidate regresses throughput. Preserving a short commit window and diff --git a/docs/pr67-publication-admission-measurement.md b/docs/pr67-publication-admission-measurement.md new file mode 100644 index 00000000..1f4aea84 --- /dev/null +++ b/docs/pr67-publication-admission-measurement.md @@ -0,0 +1,149 @@ +# PR 67: publication admission handoff and write measurement + +**ACK availability and draining improve in this diagnostic; write performance +regresses and parity remains unmet. PR #67 stays a draft.** The corrected runtime +completes 451.70 Fleet writes/s versus 481.80 before and 1,996.33 for fresh celld. +Successful scheduled p99 worsens 285.52→1,285.01 ms. All 47,471 candidate ACKs +pass warm/cold mutation and original retry audits; fleet drain takes 47.68 seconds. +One short pair establishes no repeatable causal performance change. + +## Failure and implementation + +The preceding native pipeline already moves publication-capacity waits before +global ordering, pipelines eight original follower rounds, groups canonical +follower append/fsync and reuses encoder-checked metadata after a complete fresh +origin match. Its [measurements](pr67-native-pipeline-measurement.md) show that +expensive historical/base verification and checkpoint work remain. + +A warning-only reproduction at `f858ed3cdab6699cd9c5e837ba7025c9b36f80b4` +identifies a separate post-commit failure. Two measured requests on Cells 1294 +and 1405 return `outcome_unknown`; both owner warnings preserve the original +`Capacity("pending publication bytes")`. These exact Cell IDs account for all +38 warm audit HTTP 503s. The original owner process wait exceeds 120 seconds +while drain logs `PendingPublication`. Unavailability does not establish data loss. + +After SQL has committed and returned its physical capture, the actor previously +tried a fresh global RAM reservation while retaining the command's declared +maximum result allowance. The SQL example admits up to 1 MiB of result data even +for a tiny actual result. Concurrent retained work can fill the ledger before +that second reservation; ingress headroom cannot guarantee it succeeds. + +`cf4785c675698afe786f566242d6d3dace32e775` transfers unused credit from the +original completed command to the full publication charge when it fits. The +input and actual result remain charged until their command drops; the capture +retains its own reservation until publication releases it. The total ledger +charge never increases during transfer. Larger cuts still require their entire +fresh charge under the original limit. Ordinary commands and native SQL groups +use the same handoff; group members retain their own original result sizes. +Opaque failed jobs keep their original allowance. Cancellation/deadline paths +retain admissions until the original worker actually exits. + +This fixes a demonstrated admission race without raising the 64-MiB retained, +1-GiB disk or 20-MiB publication-working budgets. It does not remove publication +I/O or promise capacity for every possible capture size. + +## Write results + +Same profile: 2,000 uniform Cells, 96-byte SQL values and durable request/result +ledger, one owner plus two followers, WAL NORMAL/tmpfs, 128 clients and queue +slots, 2,000 offered writes/s, 30-second warmup and 60-second measured window. +The before and after builds have byte-identical driver and auditor binaries, +fixture sources and images, distinct serving binaries and recorded source +identities. Celld `f2bf648663a610eefde71f3547ad61e9b896b1f0` runs fresh after +the candidate. Our builds, contributor suites and independent replay do not +overlap the timed windows. + +| Arm | Successful writes/s | Successful scheduled p99 ms | Errors | Dropped offers | +| --- | ---: | ---: | ---: | ---: | +| Cellule before, `f858ed3` | 481.80 | 285.52 | 66,660 | 24,296 | +| Cellule handoff, `cf4785c` | 451.70 | 1,285.01 | 68,219 | 24,578 | +| Fresh celld, `f2bf648` | 1,996.33 | 107.12 | 0 | 155 | + +TPS counts successful completions inside the measured window. Scheduled p99 +counts measured successful offers through client drain and excludes failures; +it is not the all-attempt histogram. Successful request p99 is 166.59, 651.04 +and 79.79 ms respectively. Candidate throughput is 6.25% lower and successful +scheduled p99 is 350.06% higher in this pair. Even restricting candidate +successes to in-window completions gives 347.97-ms p99, versus 281.77 before; +that restricted figure does not replace the qualification metric. + +Independent replay reconciles all original successful outputs, payload bytes, +per-Cell counts, offers, attempts, trailing completions and complete ACK +provenance. All 2,000 Cells have measured successes in every arm. Candidate +counts are 27,102 in-window successes and 101 trailing successes. All 68,219 +measured errors are `unavailable`, with zero `outcome_unknown`; warmup has +41,727 errors and six dropped offers. Before has 66,658 measured `unavailable` +and two `outcome_unknown`, plus 30,150 warmup errors and 6,857 warmup drops. +Celld has 1,730 warmup drops. No arm passes the unchanged performance gates. + +| Arm | Complete ACK cohort | Warm audit | Cold audit / fleet drain | +| --- | ---: | --- | --- | +| Before | 54,038 | 38 HTTP 503s; 54,000 retries checked | Cold not reached; owner wait exceeds 120 s | +| Handoff | 47,471 | All mutations and original retries pass | All pass; 47.68 s | +| Fresh celld | 180,116 | All mutations and original retries pass | All pass; 22.74 s | + +The candidate records no owner-fence warning. All 2,000 original Cells report +Idle after drain; owner and both followers exit successfully without OOM. +The passed audits verify the complete ACK cohort, including setup and warmup. +This is one successful recovery/drain observation, not qualification of every +failed-owner, transfer or collection schedule. + +## Remaining bottleneck + +The candidate's 27,175 completed native submissions have identical counts in +all seven phases, whose nanosecond totals reconcile exactly. Mean ordered-lock +wait is 0.00106 ms, publication-slot wait 0.00037 ms and total native submission +0.20103 ms. Fleet-proof wait is 53.58 ms in a separate 27,129-event cohort. +Follower window counts average 12.50 frames per sync on each member. Caller +waits overlap and these cohorts are not a serial TPS service-time partition. + +Publication debt grows during the measured window: pending entries 4,335→4,382, +oldest age 45.05→55.04 seconds, retained captures 4.20→7.84 MB and unpublished +native-log bytes 19.38→37.01 MB. Retained runtime memory rises 48.63→63.54 MB; +active Cells stay at 2,000. The window records 102,880 immutable GETs and +73,784 node-authority range starts. Removing a global admission wait and avoiding +owner fencing has not made that publication work sustainable at the offered load. + +The next architectural step remains a bounded authenticated append/lookup and +checkpoint representation that avoids rereading previous roots/history for every +selection while preserving exact original range, root and cold-recovery proofs. +It needs a deterministic work-count regression, unchanged corruption/recovery +checks and a fresh paired TPS/latency measurement. Increasing queue or memory +limits cannot establish that improvement. Larger follower shipping concurrency +also needs measured group-commit and latency evidence under original byte bounds. + +## Verification and limits + +The new real-follower regression holds publication after SQL returns a capture, +fills the entire remaining node RAM budget, then resumes the original request. +Before, it reproduces `OutcomeUnknown` with `Capacity("pending publication bytes")`. +After, three repetitions obtain a real Fleet ACK, resolve the original retry +without rerunning SQL, release all charges after joined shutdown and cold-restore +the exact published mutation/outcome. Three unit tests verify independent credit +lifetimes, oversized-cut fallback and invalid-result retention. All 41 durability +tests pass; the existing group test now checks original inputs plus actual +completed replies after unused allowances are released. Worker-cancellation +ownership assertions remain unchanged. + +The first broad suite fails the existing eight-second stalled-read assertion +before that test issues a mutation. Three unchanged-before and three unchanged- +after isolated probes pass. They support a timing/load hypothesis without proving +its cause. No read code or deadline changes. The final controlled isolated rerun +passes all 13 contributor routes: 1,993 reported workspace test/doctest executions, +38 ignored environment tests, 60 local LTX tests, Rust 1.97/1.99 Clippy, targets, +rustdoc and boundary/layout/document/contract/script gates. All failed attempts +remain recorded; no qualification profile or deadline is relaxed. + +The shared Docker VM has eight CPUs and 8,306,286,592 total memory bytes across +all roles; it does not qualify a dedicated 8-vCPU/16-GiB owner. tmpfs does not +qualify physical-media durability. Native peer protocols differ: Cellule uses +pinned mTLS and signed protobuf with a reused client per member; celld's fixture +uses native internal HTTP on loopback. This run cannot attribute the throughput +gap to that difference. No new Bucket, read-only, mixed or physical-media result +is claimed. Three paired repetitions of at least five minutes and the unchanged +zero-error/drop, latency, recovery, read and debt gates remain required. + +Raw sources, builds, journals, replay, audits, telemetry and passing/failed +verification attempts stay outside Git in +`/Volumes/Workspace/crabbuild-target/native-availability-20261009`. +Only this concise report and implementation/regression tests enter the repository. From 21029b5801a161d193c820714f6069c0820efee8 Mon Sep 17 00:00:00 2001 From: forhappy Date: Fri, 9 Oct 2026 14:37:14 -0700 Subject: [PATCH 090/102] Coalesce sparse catalog shards within phase byte admission --- .../docs/write-performance-design.md | 32 ++++-- .../src/node/bundle/index/io/metadata/mod.rs | 6 +- .../node/bundle/index/io/metadata/tests.rs | 84 ++++++++++++-- .../src/node/bundle/index/io/mod.rs | 28 +++-- .../bundle/tests/index/metadata_sparse.rs | 107 ++++++++++++++++++ .../src/node/bundle/tests/index/mod.rs | 1 + 6 files changed, 229 insertions(+), 29 deletions(-) create mode 100644 crates/cellule-runtime/src/node/bundle/tests/index/metadata_sparse.rs diff --git a/crates/cellule-runtime/docs/write-performance-design.md b/crates/cellule-runtime/docs/write-performance-design.md index 30b9917d..0a2937f9 100644 --- a/crates/cellule-runtime/docs/write-performance-design.md +++ b/crates/cellule-runtime/docs/write-performance-design.md @@ -156,20 +156,34 @@ bodies beyond the operation. Working admission remains 20 MiB; workload retention and protocol bounds remain unchanged. This is a structural charge, not allocator-profile qualification. -Catalog loading coalesces contiguous same-object shard extents after the -complete selected-shard byte preflight. Requested detached histories use a -separate phase after the shard jobs join and the combined metadata preflight -passes. History windows may bridge gaps of at most 32 KiB, charging every gap -once against the unused original 4-MiB selected-metadata allowance. Shard -windows spend no gap credit. Exhausted credit preserves separate valid reads. -Each phase joins at most eight windows before canonical decoding of every -original authenticated extent. Padding confers no authority; unrelated -histories remain references. Planning retains at most 256 shard and 4,096 +Catalog loading coalesces selected same-object shard extents after the +complete selected-shard byte preflight. Shard windows charge each bridged gap +once against that phase's unused original 4-MiB raw-body allowance; all requested +bytes and padding in the phase stay within that bound. Only the requested authenticated extents are decoded. All +shard jobs and their bodies join/drop before requested detached histories use +the separate history phase; the combined encoded shard/history preflight stays +at 4 MiB. History windows retain their 32-KiB maximum gap and spend their +existing unused selected-metadata allowance. +Exhausted gap credit preserves separate valid reads. Each phase joins at most +eight windows before canonical decoding of every original extent. Padding +confers no authority; unrelated histories remain references. Overfetch can +increase total transferred bytes across the two disjoint phases; measure bytes +as well as requests and application throughput. Planning retains at most 256 shard and 4,096 history indices as compact `u16` values, plus eight window descriptors within the existing structural working charge. No availability cache, producer reservation increase or protocol change is introduced. Coalescing can trade more transferred bytes for fewer requests; measure both costs. +A separate sparse 64-Cell preparation against 2,000 binding rows now needs +two fresh metadata reads versus 49 before shard windowing. Three fixed +repetitions cold-restore every participating Cell and its exact outcomes; all +101 bundle tests pass, including missing/corrupt origin, range and cancellation +checks. The initial history-gap-limited trial needs 11 reads and fails the new +five-read regression bound; the final shard phase uses its full original byte +credit while history keeps its prior gap limit. These component observations +establish no application TPS or latency gain; transferred bytes and end-to-end +qualification still need measurement. + The real 64-Cell, 2,000-binding preparation regression needs 123 metadata reads on the unchanged loader and three after coalescing. Three repetitions reproduce each result; every participating Cell cold-restores its exact seed diff --git a/crates/cellule-runtime/src/node/bundle/index/io/metadata/mod.rs b/crates/cellule-runtime/src/node/bundle/index/io/metadata/mod.rs index 8b250b09..c9a31184 100644 --- a/crates/cellule-runtime/src/node/bundle/index/io/metadata/mod.rs +++ b/crates/cellule-runtime/src/node/bundle/index/io/metadata/mod.rs @@ -45,6 +45,7 @@ pub(super) fn cohort<'a>( indices: &[u16], next: &mut usize, padding: &mut u64, + maximum_gap: u64, extent: impl Fn(u16) -> Result<&'a Locator>, ) -> Result> { let mut windows = Vec::with_capacity(READ_CONCURRENCY); @@ -63,10 +64,7 @@ pub(super) fn cohort<'a>( while let Some(index) = indices.get(*next) { let candidate = extent(*index)?; let gap = candidate.offset.saturating_sub(range.end); - if candidate.object != Some(object) - || gap > *padding - || gap > history::MAX_HISTORY_BYTES - { + if candidate.object != Some(object) || gap > *padding || gap > maximum_gap { break; } let end = candidate diff --git a/crates/cellule-runtime/src/node/bundle/index/io/metadata/tests.rs b/crates/cellule-runtime/src/node/bundle/index/io/metadata/tests.rs index ad118064..ef95626b 100644 --- a/crates/cellule-runtime/src/node/bundle/index/io/metadata/tests.rs +++ b/crates/cellule-runtime/src/node/bundle/index/io/metadata/tests.rs @@ -23,7 +23,14 @@ fn sparse_metadata_windows_charge_gaps_once_and_preserve_all_original_indices() sort(&mut indices, locate).unwrap(); let mut next = 0; let mut padding = 10; - let windows = cohort(&indices, &mut next, &mut padding, locate).unwrap(); + let windows = cohort( + &indices, + &mut next, + &mut padding, + history::MAX_HISTORY_BYTES, + locate, + ) + .unwrap(); assert_eq!(windows.len(), 3); assert_eq!(windows[0].indices, 0..3); assert_eq!( @@ -54,16 +61,28 @@ fn eight_metadata_windows_and_compact_protocol_indices_stay_bounded() { let mut next = 0; let mut padding = MAX_BUNDLE_BYTES; assert_eq!( - cohort(&indices, &mut next, &mut padding, locate) - .unwrap() - .len(), + cohort( + &indices, + &mut next, + &mut padding, + history::MAX_HISTORY_BYTES, + locate, + ) + .unwrap() + .len(), 8 ); assert_eq!(next, 8); assert_eq!( - cohort(&indices, &mut next, &mut padding, locate) - .unwrap() - .len(), + cohort( + &indices, + &mut next, + &mut padding, + history::MAX_HISTORY_BYTES, + locate, + ) + .unwrap() + .len(), 4 ); assert_eq!(padding, MAX_BUNDLE_BYTES); @@ -82,9 +101,58 @@ fn exhausted_gap_credit_keeps_valid_metadata_as_separate_reads() { sort(&mut indices, locate).unwrap(); let mut next = 0; let mut padding = 0; - let windows = cohort(&indices, &mut next, &mut padding, locate).unwrap(); + let windows = cohort( + &indices, + &mut next, + &mut padding, + history::MAX_HISTORY_BYTES, + locate, + ) + .unwrap(); assert_eq!(windows.len(), 2); let mut invalid = [extent(1, 0, 1)]; invalid[0].offset = u64::MAX; assert!(sort(&mut [0], |index| Ok(&invalid[usize::from(index)])).is_err()); } + +#[test] +fn sparse_shards_spend_phase_credit_while_history_retains_its_gap_limit() { + let extents = [extent(1, 0, 100), extent(1, 128 * 1024, 100)]; + let indices = [0, 1]; + let locate = |index: u16| Ok(&extents[usize::from(index)]); + let mut next = 0; + let original_credit = MAX_BUNDLE_BYTES - 200; + let mut padding = original_credit; + let shards = cohort(&indices, &mut next, &mut padding, MAX_BUNDLE_BYTES, locate).unwrap(); + assert_eq!(shards.len(), 1); + let gap = 128 * 1024 - 100; + assert_eq!(padding, original_credit - gap); + let wire = shards[0].range.end - shards[0].range.start; + assert_eq!(wire, 200 + gap); + assert!(wire <= MAX_BUNDLE_BYTES); + assert_eq!(shards[0].indices, 0..2); + let mut next = 0; + let mut padding = original_credit; + let histories = cohort( + &indices, + &mut next, + &mut padding, + history::MAX_HISTORY_BYTES, + locate, + ) + .unwrap(); + assert_eq!(histories.len(), 2); + assert_eq!( + padding, original_credit, + "history cannot spend this wide gap" + ); + let mut next = 0; + let mut padding = gap - 1; + assert_eq!( + cohort(&indices, &mut next, &mut padding, MAX_BUNDLE_BYTES, locate) + .unwrap() + .len(), + 2, + "insufficient original credit preserves separate valid shard reads" + ); +} diff --git a/crates/cellule-runtime/src/node/bundle/index/io/mod.rs b/crates/cellule-runtime/src/node/bundle/index/io/mod.rs index 14b6578f..73133586 100644 --- a/crates/cellule-runtime/src/node/bundle/index/io/mod.rs +++ b/crates/cellule-runtime/src/node/bundle/index/io/mod.rs @@ -162,9 +162,11 @@ async fn load_inner( loaded.insert(id, Vec::new()); } } - // Selected shard bytes were preflighted before any read. Shard windows - // spend no gap credit: requested history sizes are still unknown here. - // Up to eight joined windows retain the same raw-body byte allowance. + // Selected shard bytes were preflighted before any read. Sparse windows + // spend only this phase's unused raw-body allowance; decode/authenticate + // the original requested extents, never the intervening padding. Every + // shard body joins/drops before history I/O, so no padding buffer overlaps + // that phase. The combined encoded metadata preflight remains unchanged. let mut shards = Vec::with_capacity(SHARDS); for (id, shard) in root.shards.iter().enumerate() { if shard.is_some() && !wanted.is_some_and(|wanted| !wanted.contains(&(id as u8))) { @@ -180,9 +182,15 @@ async fn load_inner( }; metadata::sort(&mut shards, shard_extent)?; let mut next = 0; - let mut padding = 0; + let mut padding = MAX_BUNDLE_BYTES - selected_bytes; while next < shards.len() { - let windows = metadata::cohort(&shards, &mut next, &mut padding, shard_extent)?; + let windows = metadata::cohort( + &shards, + &mut next, + &mut padding, + MAX_BUNDLE_BYTES, + shard_extent, + )?; let bodies = try_join_all( windows .into_iter() @@ -306,9 +314,13 @@ async fn hydrate_histories( // Only eight window descriptors and at most MAX_BINDINGS compact u16 // indices are retained. Gaps charge the unused original shard/history // aggregate; this phase joins after all shard bodies have dropped. - let windows = metadata::cohort(&targets, &mut next, &mut padding, |index| { - history_extent(bindings, histories, index) - })?; + let windows = metadata::cohort( + &targets, + &mut next, + &mut padding, + history::MAX_HISTORY_BYTES, + |index| history_extent(bindings, histories, index), + )?; let bodies = try_join_all( windows .into_iter() diff --git a/crates/cellule-runtime/src/node/bundle/tests/index/metadata_sparse.rs b/crates/cellule-runtime/src/node/bundle/tests/index/metadata_sparse.rs new file mode 100644 index 00000000..bee038fd --- /dev/null +++ b/crates/cellule-runtime/src/node/bundle/tests/index/metadata_sparse.rs @@ -0,0 +1,107 @@ +use super::*; + +#[tokio::test] +async fn sparse_binding_shards_share_bounded_fresh_reads_and_exact_cold_recovery() { + let mut f = Fixture::new().await; + super::super::coverage::enroll(&mut f).await; + let mut cells = Vec::new(); + for number in 4..68 { + cells.push(f.cell(number).await); + f.heartbeat().await; + } + // Only these 64 original writers have available roots. The other rows + // populate intervening shards and grant no reconstruction capability. + inventory(&mut f, &cells[0], 1_937).await; + let head = f.node.advertisement().bundle_head().unwrap(); + let path = f.layout.node_coverage_bundle_path( + head_session(&f).as_bytes(), + head.epoch, + head.digest.as_bytes(), + ); + let mut frames = Vec::new(); + let mut assigned = Vec::new(); + for cell in &mut cells { + let (_, capture, range) = f.append(cell, 2); + frames.extend(capture); + assigned.push(range); + } + let now = f.node.advertisement().issued_at_ms(); + f.count.reset(); + let prepared = f + .directory + .prepare_node_bundle(&f.node, &frames, &assigned, now) + .await + .unwrap(); + let reads = f + .count + .requests() + .iter() + .filter(|request| request.location == path.as_ref()) + .count(); + eprintln!("sparse 2,000-binding preparation: cells=64 reads={reads}"); + assert!( + reads <= 5, + "shared sparse shards should use at most four bounded windows plus the fresh header; got {reads}" + ); + let (_, proofs) = f + .directory + .select_node_bundle(&f.node, &prepared, &f.lease, Limits::default(), now) + .await + .unwrap(); + assert_eq!(proofs.len(), cells.len()); + for proof in &proofs { + assert_eq!(proof.commit_sequence(), 2); + let cell = cells + .iter() + .find(|cell| cell.control.value().bundle_binding == Some(proof.binding())) + .unwrap(); + let overlay = proof + .recovery_overlay(&f.layout, Limits::default()) + .await + .unwrap(); + let recovered = cell + .replica + .prepare_recovered_overlay(&overlay, 1) + .await + .unwrap(); + let destination = f.scratch.path().join(format!( + "sparse-metadata-{}.sqlite", + cell.control.value().cell.as_bytes()[0] + )); + cell.replica + .open_root(&recovered.root()) + .await + .unwrap() + .restore(&destination) + .await + .unwrap(); + let db = rusqlite::Connection::open(destination).unwrap(); + let outcomes = db + .prepare("SELECT request, result FROM outcomes ORDER BY request") + .unwrap() + .query_map([], |row| { + Ok((row.get::<_, String>(0)?, row.get::<_, String>(1)?)) + }) + .unwrap() + .collect::>>() + .unwrap(); + assert_eq!( + outcomes, + vec![ + ("request-2".into(), "result-2".into()), + ("seed".into(), "original".into()) + ] + ); + } + // This call must reopen origin; neither an earlier preparation nor the + // selected proof can replace missing original metadata in a later call. + f.count.block_body_reads_for(&path); + let puts = f.count.put_requests(); + assert!( + f.directory + .prepare_node_bundle(&f.node, &frames, &assigned, now) + .await + .is_err() + ); + assert_eq!(f.count.put_requests(), puts); +} diff --git a/crates/cellule-runtime/src/node/bundle/tests/index/mod.rs b/crates/cellule-runtime/src/node/bundle/tests/index/mod.rs index 158abc16..ef57f257 100644 --- a/crates/cellule-runtime/src/node/bundle/tests/index/mod.rs +++ b/crates/cellule-runtime/src/node/bundle/tests/index/mod.rs @@ -49,4 +49,5 @@ mod encoding; mod history_cohort; mod inventory; mod metadata_cohort; +mod metadata_sparse; mod metadata_windows; From d70530c9a004d0b24199c52db5253908033be14f Mon Sep 17 00:00:00 2001 From: forhappy Date: Fri, 9 Oct 2026 15:11:10 -0700 Subject: [PATCH 091/102] Revert "Coalesce sparse catalog shards within phase byte admission" This reverts commit 21029b5801a161d193c820714f6069c0820efee8. --- .../docs/write-performance-design.md | 32 ++---- .../src/node/bundle/index/io/metadata/mod.rs | 6 +- .../node/bundle/index/io/metadata/tests.rs | 84 ++------------ .../src/node/bundle/index/io/mod.rs | 28 ++--- .../bundle/tests/index/metadata_sparse.rs | 107 ------------------ .../src/node/bundle/tests/index/mod.rs | 1 - 6 files changed, 29 insertions(+), 229 deletions(-) delete mode 100644 crates/cellule-runtime/src/node/bundle/tests/index/metadata_sparse.rs diff --git a/crates/cellule-runtime/docs/write-performance-design.md b/crates/cellule-runtime/docs/write-performance-design.md index 0a2937f9..30b9917d 100644 --- a/crates/cellule-runtime/docs/write-performance-design.md +++ b/crates/cellule-runtime/docs/write-performance-design.md @@ -156,34 +156,20 @@ bodies beyond the operation. Working admission remains 20 MiB; workload retention and protocol bounds remain unchanged. This is a structural charge, not allocator-profile qualification. -Catalog loading coalesces selected same-object shard extents after the -complete selected-shard byte preflight. Shard windows charge each bridged gap -once against that phase's unused original 4-MiB raw-body allowance; all requested -bytes and padding in the phase stay within that bound. Only the requested authenticated extents are decoded. All -shard jobs and their bodies join/drop before requested detached histories use -the separate history phase; the combined encoded shard/history preflight stays -at 4 MiB. History windows retain their 32-KiB maximum gap and spend their -existing unused selected-metadata allowance. -Exhausted gap credit preserves separate valid reads. Each phase joins at most -eight windows before canonical decoding of every original extent. Padding -confers no authority; unrelated histories remain references. Overfetch can -increase total transferred bytes across the two disjoint phases; measure bytes -as well as requests and application throughput. Planning retains at most 256 shard and 4,096 +Catalog loading coalesces contiguous same-object shard extents after the +complete selected-shard byte preflight. Requested detached histories use a +separate phase after the shard jobs join and the combined metadata preflight +passes. History windows may bridge gaps of at most 32 KiB, charging every gap +once against the unused original 4-MiB selected-metadata allowance. Shard +windows spend no gap credit. Exhausted credit preserves separate valid reads. +Each phase joins at most eight windows before canonical decoding of every +original authenticated extent. Padding confers no authority; unrelated +histories remain references. Planning retains at most 256 shard and 4,096 history indices as compact `u16` values, plus eight window descriptors within the existing structural working charge. No availability cache, producer reservation increase or protocol change is introduced. Coalescing can trade more transferred bytes for fewer requests; measure both costs. -A separate sparse 64-Cell preparation against 2,000 binding rows now needs -two fresh metadata reads versus 49 before shard windowing. Three fixed -repetitions cold-restore every participating Cell and its exact outcomes; all -101 bundle tests pass, including missing/corrupt origin, range and cancellation -checks. The initial history-gap-limited trial needs 11 reads and fails the new -five-read regression bound; the final shard phase uses its full original byte -credit while history keeps its prior gap limit. These component observations -establish no application TPS or latency gain; transferred bytes and end-to-end -qualification still need measurement. - The real 64-Cell, 2,000-binding preparation regression needs 123 metadata reads on the unchanged loader and three after coalescing. Three repetitions reproduce each result; every participating Cell cold-restores its exact seed diff --git a/crates/cellule-runtime/src/node/bundle/index/io/metadata/mod.rs b/crates/cellule-runtime/src/node/bundle/index/io/metadata/mod.rs index c9a31184..8b250b09 100644 --- a/crates/cellule-runtime/src/node/bundle/index/io/metadata/mod.rs +++ b/crates/cellule-runtime/src/node/bundle/index/io/metadata/mod.rs @@ -45,7 +45,6 @@ pub(super) fn cohort<'a>( indices: &[u16], next: &mut usize, padding: &mut u64, - maximum_gap: u64, extent: impl Fn(u16) -> Result<&'a Locator>, ) -> Result> { let mut windows = Vec::with_capacity(READ_CONCURRENCY); @@ -64,7 +63,10 @@ pub(super) fn cohort<'a>( while let Some(index) = indices.get(*next) { let candidate = extent(*index)?; let gap = candidate.offset.saturating_sub(range.end); - if candidate.object != Some(object) || gap > *padding || gap > maximum_gap { + if candidate.object != Some(object) + || gap > *padding + || gap > history::MAX_HISTORY_BYTES + { break; } let end = candidate diff --git a/crates/cellule-runtime/src/node/bundle/index/io/metadata/tests.rs b/crates/cellule-runtime/src/node/bundle/index/io/metadata/tests.rs index ef95626b..ad118064 100644 --- a/crates/cellule-runtime/src/node/bundle/index/io/metadata/tests.rs +++ b/crates/cellule-runtime/src/node/bundle/index/io/metadata/tests.rs @@ -23,14 +23,7 @@ fn sparse_metadata_windows_charge_gaps_once_and_preserve_all_original_indices() sort(&mut indices, locate).unwrap(); let mut next = 0; let mut padding = 10; - let windows = cohort( - &indices, - &mut next, - &mut padding, - history::MAX_HISTORY_BYTES, - locate, - ) - .unwrap(); + let windows = cohort(&indices, &mut next, &mut padding, locate).unwrap(); assert_eq!(windows.len(), 3); assert_eq!(windows[0].indices, 0..3); assert_eq!( @@ -61,28 +54,16 @@ fn eight_metadata_windows_and_compact_protocol_indices_stay_bounded() { let mut next = 0; let mut padding = MAX_BUNDLE_BYTES; assert_eq!( - cohort( - &indices, - &mut next, - &mut padding, - history::MAX_HISTORY_BYTES, - locate, - ) - .unwrap() - .len(), + cohort(&indices, &mut next, &mut padding, locate) + .unwrap() + .len(), 8 ); assert_eq!(next, 8); assert_eq!( - cohort( - &indices, - &mut next, - &mut padding, - history::MAX_HISTORY_BYTES, - locate, - ) - .unwrap() - .len(), + cohort(&indices, &mut next, &mut padding, locate) + .unwrap() + .len(), 4 ); assert_eq!(padding, MAX_BUNDLE_BYTES); @@ -101,58 +82,9 @@ fn exhausted_gap_credit_keeps_valid_metadata_as_separate_reads() { sort(&mut indices, locate).unwrap(); let mut next = 0; let mut padding = 0; - let windows = cohort( - &indices, - &mut next, - &mut padding, - history::MAX_HISTORY_BYTES, - locate, - ) - .unwrap(); + let windows = cohort(&indices, &mut next, &mut padding, locate).unwrap(); assert_eq!(windows.len(), 2); let mut invalid = [extent(1, 0, 1)]; invalid[0].offset = u64::MAX; assert!(sort(&mut [0], |index| Ok(&invalid[usize::from(index)])).is_err()); } - -#[test] -fn sparse_shards_spend_phase_credit_while_history_retains_its_gap_limit() { - let extents = [extent(1, 0, 100), extent(1, 128 * 1024, 100)]; - let indices = [0, 1]; - let locate = |index: u16| Ok(&extents[usize::from(index)]); - let mut next = 0; - let original_credit = MAX_BUNDLE_BYTES - 200; - let mut padding = original_credit; - let shards = cohort(&indices, &mut next, &mut padding, MAX_BUNDLE_BYTES, locate).unwrap(); - assert_eq!(shards.len(), 1); - let gap = 128 * 1024 - 100; - assert_eq!(padding, original_credit - gap); - let wire = shards[0].range.end - shards[0].range.start; - assert_eq!(wire, 200 + gap); - assert!(wire <= MAX_BUNDLE_BYTES); - assert_eq!(shards[0].indices, 0..2); - let mut next = 0; - let mut padding = original_credit; - let histories = cohort( - &indices, - &mut next, - &mut padding, - history::MAX_HISTORY_BYTES, - locate, - ) - .unwrap(); - assert_eq!(histories.len(), 2); - assert_eq!( - padding, original_credit, - "history cannot spend this wide gap" - ); - let mut next = 0; - let mut padding = gap - 1; - assert_eq!( - cohort(&indices, &mut next, &mut padding, MAX_BUNDLE_BYTES, locate) - .unwrap() - .len(), - 2, - "insufficient original credit preserves separate valid shard reads" - ); -} diff --git a/crates/cellule-runtime/src/node/bundle/index/io/mod.rs b/crates/cellule-runtime/src/node/bundle/index/io/mod.rs index 73133586..14b6578f 100644 --- a/crates/cellule-runtime/src/node/bundle/index/io/mod.rs +++ b/crates/cellule-runtime/src/node/bundle/index/io/mod.rs @@ -162,11 +162,9 @@ async fn load_inner( loaded.insert(id, Vec::new()); } } - // Selected shard bytes were preflighted before any read. Sparse windows - // spend only this phase's unused raw-body allowance; decode/authenticate - // the original requested extents, never the intervening padding. Every - // shard body joins/drops before history I/O, so no padding buffer overlaps - // that phase. The combined encoded metadata preflight remains unchanged. + // Selected shard bytes were preflighted before any read. Shard windows + // spend no gap credit: requested history sizes are still unknown here. + // Up to eight joined windows retain the same raw-body byte allowance. let mut shards = Vec::with_capacity(SHARDS); for (id, shard) in root.shards.iter().enumerate() { if shard.is_some() && !wanted.is_some_and(|wanted| !wanted.contains(&(id as u8))) { @@ -182,15 +180,9 @@ async fn load_inner( }; metadata::sort(&mut shards, shard_extent)?; let mut next = 0; - let mut padding = MAX_BUNDLE_BYTES - selected_bytes; + let mut padding = 0; while next < shards.len() { - let windows = metadata::cohort( - &shards, - &mut next, - &mut padding, - MAX_BUNDLE_BYTES, - shard_extent, - )?; + let windows = metadata::cohort(&shards, &mut next, &mut padding, shard_extent)?; let bodies = try_join_all( windows .into_iter() @@ -314,13 +306,9 @@ async fn hydrate_histories( // Only eight window descriptors and at most MAX_BINDINGS compact u16 // indices are retained. Gaps charge the unused original shard/history // aggregate; this phase joins after all shard bodies have dropped. - let windows = metadata::cohort( - &targets, - &mut next, - &mut padding, - history::MAX_HISTORY_BYTES, - |index| history_extent(bindings, histories, index), - )?; + let windows = metadata::cohort(&targets, &mut next, &mut padding, |index| { + history_extent(bindings, histories, index) + })?; let bodies = try_join_all( windows .into_iter() diff --git a/crates/cellule-runtime/src/node/bundle/tests/index/metadata_sparse.rs b/crates/cellule-runtime/src/node/bundle/tests/index/metadata_sparse.rs deleted file mode 100644 index bee038fd..00000000 --- a/crates/cellule-runtime/src/node/bundle/tests/index/metadata_sparse.rs +++ /dev/null @@ -1,107 +0,0 @@ -use super::*; - -#[tokio::test] -async fn sparse_binding_shards_share_bounded_fresh_reads_and_exact_cold_recovery() { - let mut f = Fixture::new().await; - super::super::coverage::enroll(&mut f).await; - let mut cells = Vec::new(); - for number in 4..68 { - cells.push(f.cell(number).await); - f.heartbeat().await; - } - // Only these 64 original writers have available roots. The other rows - // populate intervening shards and grant no reconstruction capability. - inventory(&mut f, &cells[0], 1_937).await; - let head = f.node.advertisement().bundle_head().unwrap(); - let path = f.layout.node_coverage_bundle_path( - head_session(&f).as_bytes(), - head.epoch, - head.digest.as_bytes(), - ); - let mut frames = Vec::new(); - let mut assigned = Vec::new(); - for cell in &mut cells { - let (_, capture, range) = f.append(cell, 2); - frames.extend(capture); - assigned.push(range); - } - let now = f.node.advertisement().issued_at_ms(); - f.count.reset(); - let prepared = f - .directory - .prepare_node_bundle(&f.node, &frames, &assigned, now) - .await - .unwrap(); - let reads = f - .count - .requests() - .iter() - .filter(|request| request.location == path.as_ref()) - .count(); - eprintln!("sparse 2,000-binding preparation: cells=64 reads={reads}"); - assert!( - reads <= 5, - "shared sparse shards should use at most four bounded windows plus the fresh header; got {reads}" - ); - let (_, proofs) = f - .directory - .select_node_bundle(&f.node, &prepared, &f.lease, Limits::default(), now) - .await - .unwrap(); - assert_eq!(proofs.len(), cells.len()); - for proof in &proofs { - assert_eq!(proof.commit_sequence(), 2); - let cell = cells - .iter() - .find(|cell| cell.control.value().bundle_binding == Some(proof.binding())) - .unwrap(); - let overlay = proof - .recovery_overlay(&f.layout, Limits::default()) - .await - .unwrap(); - let recovered = cell - .replica - .prepare_recovered_overlay(&overlay, 1) - .await - .unwrap(); - let destination = f.scratch.path().join(format!( - "sparse-metadata-{}.sqlite", - cell.control.value().cell.as_bytes()[0] - )); - cell.replica - .open_root(&recovered.root()) - .await - .unwrap() - .restore(&destination) - .await - .unwrap(); - let db = rusqlite::Connection::open(destination).unwrap(); - let outcomes = db - .prepare("SELECT request, result FROM outcomes ORDER BY request") - .unwrap() - .query_map([], |row| { - Ok((row.get::<_, String>(0)?, row.get::<_, String>(1)?)) - }) - .unwrap() - .collect::>>() - .unwrap(); - assert_eq!( - outcomes, - vec![ - ("request-2".into(), "result-2".into()), - ("seed".into(), "original".into()) - ] - ); - } - // This call must reopen origin; neither an earlier preparation nor the - // selected proof can replace missing original metadata in a later call. - f.count.block_body_reads_for(&path); - let puts = f.count.put_requests(); - assert!( - f.directory - .prepare_node_bundle(&f.node, &frames, &assigned, now) - .await - .is_err() - ); - assert_eq!(f.count.put_requests(), puts); -} diff --git a/crates/cellule-runtime/src/node/bundle/tests/index/mod.rs b/crates/cellule-runtime/src/node/bundle/tests/index/mod.rs index ef57f257..158abc16 100644 --- a/crates/cellule-runtime/src/node/bundle/tests/index/mod.rs +++ b/crates/cellule-runtime/src/node/bundle/tests/index/mod.rs @@ -49,5 +49,4 @@ mod encoding; mod history_cohort; mod inventory; mod metadata_cohort; -mod metadata_sparse; mod metadata_windows; From 6450734ace4529c5d7073b0b4e7562c0539d9701 Mon Sep 17 00:00:00 2001 From: forhappy Date: Fri, 9 Oct 2026 15:15:00 -0700 Subject: [PATCH 092/102] Record rejected shard batching and fresh write comparison --- .../docs/write-performance-design.md | 12 +- .../pr67-publication-admission-measurement.md | 4 + docs/pr67-shard-window-measurement.md | 141 ++++++++++++++++++ 3 files changed, 156 insertions(+), 1 deletion(-) create mode 100644 docs/pr67-shard-window-measurement.md diff --git a/crates/cellule-runtime/docs/write-performance-design.md b/crates/cellule-runtime/docs/write-performance-design.md index 30b9917d..ad69ed25 100644 --- a/crates/cellule-runtime/docs/write-performance-design.md +++ b/crates/cellule-runtime/docs/write-performance-design.md @@ -1,6 +1,16 @@ # Node write and read performance design -The latest [publication-admission measurement](../../../docs/pr67-publication-admission-measurement.md) +The latest [sparse-shard diagnostic](../../../docs/pr67-shard-window-measurement.md) +rejects and reverts wide metadata reads. Its component fixture improves 49→2 +reads, but fresh application TPS falls 510.13→300.63/s and successful scheduled +p99 rises 1,314.72→2,554.88 ms. Node-authority range bytes rise 61.17% despite +fewer requests. Fresh celld completes 1,999.82/s with 14.06-ms p99 and zero +errors/drops. The prototype passes all 44,314 warm/cold ACK checks; the retained +baseline fails 110 warm checks with HTTP 503s. Prioritize total publication +bytes/work and warm availability under backlog. No arm is qualified; the +acceptance gates below remain unchanged. + +The preceding [publication-admission measurement](../../../docs/pr67-publication-admission-measurement.md) identifies a post-SQL RAM reservation race and transfers unused original command credit to publication. All 47,471 candidate ACKs pass warm/cold mutation and original retry audits; fleet drain takes 47.68 seconds. Write performance diff --git a/docs/pr67-publication-admission-measurement.md b/docs/pr67-publication-admission-measurement.md index 1f4aea84..c2b199a2 100644 --- a/docs/pr67-publication-admission-measurement.md +++ b/docs/pr67-publication-admission-measurement.md @@ -1,5 +1,9 @@ # PR 67: publication admission handoff and write measurement +The newer [sparse-shard diagnostic](pr67-shard-window-measurement.md) rejects +wide metadata reads after a fresh TPS/latency regression. Its retained-runtime +baseline also fails warm ACK availability; those results are a separate pair. + **ACK availability and draining improve in this diagnostic; write performance regresses and parity remains unmet. PR #67 stays a draft.** The corrected runtime completes 451.70 Fleet writes/s versus 481.80 before and 1,996.33 for fresh celld. diff --git a/docs/pr67-shard-window-measurement.md b/docs/pr67-shard-window-measurement.md new file mode 100644 index 00000000..74eb80cd --- /dev/null +++ b/docs/pr67-shard-window-measurement.md @@ -0,0 +1,141 @@ +# PR 67: rejected sparse shard batching diagnostic + +**The wide-read prototype is reverted. PR #67 remains a draft; performance parity +is unmet.** A fresh pair completes 510.13 Fleet writes/s for the retained runtime, +300.63/s for the prototype and 1,999.82/s for celld. Successful scheduled p99 is +1,314.72, 2,554.88 and 14.06 ms respectively. Fewer metadata requests did not +demonstrate an application performance benefit. One short pair establishes no +repeatable causal attribution. + +## Experiment and disposition + +The retained [native pipeline](pr67-native-pipeline-measurement.md) moves +publication admission before global ordering, pipelines eight original follower +rounds and uses follower group commit. It reuses private encoder-checked metadata +only after a complete fresh origin match. The [command-credit handoff](pr67-publication-admission-measurement.md) +also remains. Repeated base/history verification and checkpoint work are still +expensive; these existing changes do not establish sustainable publication. + +Prototype `21029b5801a161d193c820714f6069c0820efee8` lets requested catalog shards +share sparse object ranges using the unused 4-MiB shard-phase raw-body allowance. +It authenticates only requested extents; padding grants no authority. The +combined selected encoded metadata preflight remains 4 MiB, history gaps retain +their 32-KiB limit and shard bodies join/drop before history I/O. Working memory, +concurrency, persisted formats and fresh dependency checks are unchanged. + +The controlled 2,000-binding/64-Cell fixture falls from 49 reads to two in three +repetitions, with exact cold mutation and request-result recovery for all 64 +Cells. Later unavailable original metadata still rejects preparation without a +new PUT. An initial 32-KiB-gap trial takes 11 reads and fails the original +at-most-five-read assertion; that assertion is never loosened. All 101 bundle +tests pass on the final prototype, but its application performance fails below. + +Revert `d70530c9a004d0b24199c52db5253908033be14f` restores the production source +byte-identically to the preceding PR head `1f3d78d`. The new fixture, private +planner test and prototype source remain in external evidence; the rejected +optimization is absent from the delivered implementation. This is a measured +rejection, not a new performance improvement. + +## Fresh application results + +Same SQL ledger workload: 2,000 uniformly active Cells, 96-byte values, one owner +and two followers, WAL NORMAL/tmpfs, 128 clients and queue slots, 2,000 offered +writes/s, 30-second warmup and 60-second measured window. Baseline `cf4785c` +and prototype `21029b5` have byte-identical driver/auditor binaries, fixtures, +workload and pinned images, with recorded distinct serving binaries. Fresh +celld is `f2bf648663a610eefde71f3547ad61e9b896b1f0`. Builds, contributor checks +and independent journal replay do not overlap the timed windows. + +| Arm | Successful writes/s | Successful scheduled p99 ms | Errors | Dropped offers | +| --- | ---: | ---: | ---: | ---: | +| Retained Cellule, `cf4785c` | 510.13 | 1,314.72 | 61,426 | 27,966 | +| Rejected prototype, `21029b5` | 300.63 | 2,554.88 | 15,172 | 86,607 | +| Fresh celld, `f2bf648` | 1,999.82 | 14.06 | 0 | 0 | + +TPS counts successful completions inside the measured window. Scheduled p99 +includes measured successful offers through client drain and excludes failures; +it is not the all-attempt histogram. Successful request p99 is 532.86, 1,418.80 +and 12.92 ms respectively. Prototype TPS falls 41.07% and scheduled p99 rises +94.33% in this pair. Its 183 trailing successes do not count toward in-window +TPS. Celld has 11 trailing successes. Every arm has measured successes on all +2,000 Cells. + +Independent replay reconciles original successful outputs, payload bytes, +per-Cell counts, every planned/generated offer and attempt, trailing completions +and the complete ACK cohort. All measured and warmup Cellule errors are +`unavailable`, with zero `outcome_unknown`. Baseline warmup has 32,968 errors +and 2,622 drops; prototype warmup has 16,814 errors and 19,094 drops. Celld has +zero warmup errors or drops. + +| Arm | Complete ACKs | Warm mutation / original retry audit | Cold audit / fleet drain | +| --- | ---: | --- | --- | +| Retained Cellule | 57,019 | 110 HTTP 503s; 56,909 retries checked | Cold not reached; timed drain result absent | +| Rejected prototype | 44,314 | All pass | All pass; 46.48 s | +| Fresh celld | 182,001 | All pass | All pass; 14.24 s | + +Both Cellule owner logs contain zero fencing warnings and all 2,000 Cells report +Idle during original cleanup/drain. All original owner/follower containers exit +successfully without OOM. Baseline audit unavailability does not establish lost +data; it is still a failed availability/recovery gate. The prototype's passed +cohort audit does not qualify every failed-owner, transfer or collection schedule. + +## Publication cost and next work + +| Measured-window metric | Retained Cellule | Rejected prototype | +| --- | ---: | ---: | +| Node-authority range starts | 89,386 | 75,710 | +| Node-authority range bytes | 985,951,049 | 1,589,052,493 | +| Range bytes / successful in-window write | 32,212 | 88,095 | +| Immutable GET starts / bytes | 105,383 / 386,976,745 | 59,337 / 206,666,996 | +| Mean ordered-lock wait ms | 0.00068 | 0.00068 | +| Mean native submission ms | 0.14479 | 0.64885 | +| Separate mean Fleet-proof wait ms | 45.96 | 121.57 | +| Follower frames / sync, two members | 12.12 / 12.11 | 8.94 / 8.96 | + +Native submissions have identical counts across all seven phases and their +nanosecond totals reconcile exactly in each arm. The Fleet-proof cohort is +separate and caller waits overlap; these means are not a serial TPS partition. +Range starts fall 15.30% but transferred range bytes rise 61.17%, even while +successful output falls. This supports the byte-amplification concern without +proving that one factor causes the entire regression. + +Baseline unpublished native bytes grow 28.17→45.86 MB and retained runtime +memory 54.12→64.10 MB. Prototype native debt grows 33.86→42.57 MB and oldest +unpublished age 53.22→97.47 seconds. Its pending entries fall 5,632→2,135 and +memory 62.90→37.93 MB, which does not establish stable publication: native debt +and age still grow. Active Cell count remains 2,000 in both arms. + +Prioritize a bounded authenticated append/lookup and checkpoint representation +that reduces total requests **and bytes per verified issued range**. Preserve +fresh origin/dependency checks and complete suffix/cold-recovery proofs; measure +checkpoint work separately from shared selection. Diagnose warm audit admission +under backlog with original errors and deadlines. Keep the existing ordering +and follower pipeline bounds until new evidence justifies changing them. Each +next candidate needs an unchanged corruption/recovery regression and a fresh +paired application measurement, not only a component read-count reduction. + +## Verification and limits + +The prototype passes all 13 contributor routes in an isolated snapshot: 1,995 +reported workspace test/doctest executions (including child-process reporting), +38 ignored environment tests, 60 local LTX tests, Rust 1.97/1.99 Clippy, +targets, rustdoc and boundary/layout/document/contract/script gates. Its three +sparse fixture repetitions and 101 bundle tests also pass. The invalid initial +cache-overlap compilation attempt, controlled 49-read baseline failure and +11-read prototype failure are preserved, not counted as passing evidence. +The restored production source is identical to the preceding verified head; +final document gates cover only the report/design edits. + +All canonical qualification reports remain false. The shared Docker VM has +eight CPUs and 8,306,286,592 total memory bytes across all roles; it does not +qualify a dedicated 8-vCPU/16-GiB owner. tmpfs does not qualify physical-media +durability. Cellule uses pinned mTLS/signed protobuf with a reused client per +member; celld's native fixture uses internal HTTP on loopback. These protocol +differences remain, and no transport-cost attribution is claimed. No new Bucket, +read-only or mixed result is claimed. Three matched repetitions of at least +five minutes and unchanged zero-error/drop, latency, recovery, read and debt +gates remain required. + +Raw sources, builds, complete journals, audits, telemetry and passing/failed +attempts stay outside Git at +`/Volumes/Workspace/crabbuild-target/native-shard-window-20261009`. From 634bd90e0a3415301f02b922d65e8e4c52a6db55 Mon Sep 17 00:00:00 2001 From: forhappy Date: Fri, 9 Oct 2026 15:32:28 -0700 Subject: [PATCH 093/102] Verify checkpoint bases in the admitted publication cohort --- .../docs/write-performance-design.md | 9 +- .../src/node/bundle/closure.rs | 28 ++- .../bundle/tests/index/checkpoint_cohort.rs | 181 ++++++++++++++++++ .../src/node/bundle/tests/index/mod.rs | 1 + .../src/node/bundle/verification/base.rs | 8 +- .../src/node/bundle/verification/mod.rs | 6 +- 6 files changed, 223 insertions(+), 10 deletions(-) create mode 100644 crates/cellule-runtime/src/node/bundle/tests/index/checkpoint_cohort.rs diff --git a/crates/cellule-runtime/docs/write-performance-design.md b/crates/cellule-runtime/docs/write-performance-design.md index ad69ed25..f2a45d45 100644 --- a/crates/cellule-runtime/docs/write-performance-design.md +++ b/crates/cellule-runtime/docs/write-performance-design.md @@ -146,7 +146,14 @@ Startup admission precedes installation of the irreversible feed. Selection and exact root checkpoints use the same original binding/heartbeat authority; 512 checkpoint requests are bounded and their callbacks join before Cell closure. Fair turns alternate queued native work and -checkpoint cohorts. The Fleet SQL example installs this producer; the current +checkpoint cohorts. Exact checkpoints reuse selection's canonical fresh-base +verifier for changed participants: at most eight small-root operations with a +4-MiB operation charge, within the original producer working reservation. +Larger graphs stay serial. No catalog upload or CAS occurs before every original +participant joins verification; no prior availability proof is cached. The real +held-root/body fixture verifies this overlap, cancellation and exact cold results; +application throughput still requires a fresh measurement. +The Fleet SQL example installs this producer; the current Bucket-only performance adapter bypasses it. Each selection reads its complete new cohort object once from origin, compares diff --git a/crates/cellule-runtime/src/node/bundle/closure.rs b/crates/cellule-runtime/src/node/bundle/closure.rs index 2fcfb408..0c9e82ad 100644 --- a/crates/cellule-runtime/src/node/bundle/closure.rs +++ b/crates/cellule-runtime/src/node/bundle/closure.rs @@ -25,6 +25,8 @@ impl NodeDirectory { /// index upload and one shared node CAS. Each root must match its opaque /// complete-capture proof; any missing/stale participant leaves the catalog /// unchanged. Callers retain their admissions until this operation joins. + /// Fresh small-root verification overlaps at most eight operations within + /// the original 4-MiB working allowance; larger graphs remain serial. pub async fn checkpoint_bundle_cells( &self, observed: &VersionedNodeAdvertisement, @@ -57,7 +59,7 @@ impl NodeDirectory { let mut catalog = store::load_catalog_cells(&self.layout, observed.advertisement.session, head, &cells) .await?; - let mut changed = false; + let mut changed = Vec::with_capacity(checkpoints.len()); for (authority, proof) in checkpoints { let pin = proof.binding(); let binding = catalog.binding_mut(pin.digest)?; @@ -90,16 +92,34 @@ impl NodeDirectory { continue; } binding.control = control.clone(); - verify_base(&self.layout, binding, limits).await?; binding.locators.drain(..prefix); if binding.locators.is_empty() { binding.first_commit = binding.selected_commit; } - changed = true; + changed.push(pin.digest); } - if !changed { + if changed.is_empty() { return Ok(observed.clone()); } + { + // These are only the original changed participants, never siblings. + // Catalog edits remain private until every fresh root/dependency + // joins the same admitted verifier used by shared selection. + let bases = catalog + .bindings + .iter() + .filter(|binding| { + binding + .control + .bundle_binding + .is_some_and(|pin| changed.contains(&pin.digest)) + }) + .collect::>(); + if bases.len() != changed.len() { + return Err(Error::Node("bundle checkpoint participant is absent")); + } + verification::verify_bases(&self.layout, &bases, limits).await?; + } let prepared = self.upload_catalog(Some(head), catalog, &[]).await?; self.select_catalog(observed, &prepared, now_ms).await } diff --git a/crates/cellule-runtime/src/node/bundle/tests/index/checkpoint_cohort.rs b/crates/cellule-runtime/src/node/bundle/tests/index/checkpoint_cohort.rs new file mode 100644 index 00000000..e449820d --- /dev/null +++ b/crates/cellule-runtime/src/node/bundle/tests/index/checkpoint_cohort.rs @@ -0,0 +1,181 @@ +use super::*; +use std::sync::atomic::Ordering; + +#[tokio::test] +async fn checkpoint_fresh_roots_overlap_within_original_working_admission() { + held_checkpoint_bases(12).await; +} + +#[tokio::test] +async fn checkpoint_fresh_bodies_join_before_upload_and_exact_cold_recovery() { + held_checkpoint_bases(13).await; +} + +async fn held_checkpoint_bases(mode: u8) { + let faults = Arc::new(super::super::faults::ReplyFault::default()); + let mut f = Fixture::with_store(faults.clone()).await; + let mut cells = Vec::new(); + for number in 4..14 { + cells.push(f.cell(number).await); + f.heartbeat().await; + } + let mut frames = Vec::new(); + let mut assigned = Vec::new(); + for cell in &mut cells { + let (_, capture, range) = f.append(cell, 2); + frames.extend(capture); + assigned.push(range); + } + let now = f.node.advertisement().issued_at_ms(); + let proposal = f + .directory + .prepare_node_bundle(&f.node, &frames, &assigned, now) + .await + .unwrap(); + let (node, proofs) = f + .directory + .select_node_bundle(&f.node, &proposal, &f.lease, Limits::default(), now) + .await + .unwrap(); + f.node = node; + for proof in &proofs { + let cell = cells + .iter() + .find(|cell| cell.control.value().bundle_binding == Some(proof.binding())) + .unwrap(); + f.publisher(cell).materialize_bundle(proof).await.unwrap(); + f.heartbeat().await; + } + let checkpoints: Vec<_> = proofs + .iter() + .map(|proof| { + let cell = cells + .iter() + .find(|cell| cell.control.value().bundle_binding == Some(proof.binding())) + .unwrap(); + (&cell.authority, proof) + }) + .collect(); + f.count.reset(); + faults.mode.store(mode, Ordering::SeqCst); + let now = f.node.advertisement().issued_at_ms(); + { + let checkpoint = + f.directory + .checkpoint_bundle_cells(&f.node, &checkpoints, Limits::default(), now); + tokio::pin!(checkpoint); + let overlap = async { + loop { + let started = faults.node_started.notified(); + if faults.base_started.load(Ordering::SeqCst) >= 8 { + break; + } + started.await; + } + }; + tokio::select! { + _ = &mut checkpoint => panic!("held checkpoint bases must prevent upload and CAS"), + result = tokio::time::timeout(std::time::Duration::from_secs(5), overlap) => { + println!("checkpoint original dependencies started before release={}", faults.base_started.load(Ordering::SeqCst)); + result.unwrap(); + }, + } + assert_eq!( + faults.base_started.load(Ordering::SeqCst), + 8, + "checkpoint verification must overlap only the admitted first eight bases" + ); + } + assert_eq!(f.count.put_requests(), 0); + faults.mode.store(0, Ordering::SeqCst); + faults.node_resume.notify_waiters(); + let original = cells[0] + .authority + .load(cells[0].control.value().cell) + .await + .unwrap() + .unwrap() + .value() + .ltx_root() + .unwrap(); + let missing = f.layout.incarnation_object_path( + &original.cell, + &original.incarnation, + &original.digest, + cellule_ltx::CellObjectKind::Root, + ); + f.count.block_body_reads_for(&missing); + f.count.reset(); + assert!( + f.directory + .checkpoint_bundle_cells(&f.node, &checkpoints, Limits::default(), now) + .await + .is_err() + ); + assert_eq!(f.count.put_requests(), 0); + f.count.unblock_body_reads_for(&missing); + f.count.reset(); + f.node = f + .directory + .checkpoint_bundle_cells(&f.node, &checkpoints, Limits::default(), now) + .await + .unwrap(); + assert_eq!( + f.count + .requests() + .iter() + .filter(|request| request.location.ends_with(".root")) + .count(), + cells.len(), + "each eligible checkpoint must freshly authenticate its original root once" + ); + assert_eq!(f.count.put_requests(), 2); + for (number, cell) in cells.iter().enumerate() { + let current = cell + .authority + .load(cell.control.value().cell) + .await + .unwrap() + .unwrap(); + let selected = f + .directory + .load_bundle_coverage(&cell.authority, ¤t, Limits::default()) + .await + .unwrap(); + assert_eq!(selected.locator_count(), 0); + assert_eq!(selected.commit_sequence(), 2); + let root = selected.base().unwrap(); + let cold = CellReplica::new( + f.layout.clone(), + root.cell, + root.incarnation, + Limits::default(), + ) + .unwrap(); + let destination = f + .scratch + .path() + .join(format!("checkpoint-cohort-{number}.sqlite")); + cold.open_root(&root) + .await + .unwrap() + .restore(&destination) + .await + .unwrap(); + let db = rusqlite::Connection::open(destination).unwrap(); + let outcomes: Vec<(String, String)> = db + .prepare("SELECT request, result FROM outcomes ORDER BY request") + .unwrap() + .query_map([], |row| Ok((row.get(0)?, row.get(1)?))) + .unwrap() + .collect::>() + .unwrap(); + assert_eq!( + outcomes, + vec![ + ("request-2".into(), "result-2".into()), + ("seed".into(), "original".into()) + ] + ); + } +} diff --git a/crates/cellule-runtime/src/node/bundle/tests/index/mod.rs b/crates/cellule-runtime/src/node/bundle/tests/index/mod.rs index 158abc16..4ca166da 100644 --- a/crates/cellule-runtime/src/node/bundle/tests/index/mod.rs +++ b/crates/cellule-runtime/src/node/bundle/tests/index/mod.rs @@ -41,6 +41,7 @@ fn head_session(f: &Fixture) -> SessionId { mod base_cohort; mod bootstrap; mod checkpoint; +mod checkpoint_cohort; mod cohort; mod compatibility; mod copy_on_write; diff --git a/crates/cellule-runtime/src/node/bundle/verification/base.rs b/crates/cellule-runtime/src/node/bundle/verification/base.rs index 52a3595c..a9db0d62 100644 --- a/crates/cellule-runtime/src/node/bundle/verification/base.rs +++ b/crates/cellule-runtime/src/node/bundle/verification/base.rs @@ -1,9 +1,9 @@ //! Admitted fresh small-base verification; larger graphs stay serial. use super::*; -pub(super) async fn verify( +pub(in crate::node::bundle) async fn verify( layout: &cellule_ltx::CellStorageLayout, - bindings: &[Binding], + bindings: &[&Binding], limits: cellule_ltx::Limits, ) -> Result<()> { for cohort in bindings.chunks(READ_CONCURRENCY) { @@ -13,7 +13,7 @@ pub(super) async fn verify( .map(|index| async move { Ok::<_, Error>(( index, - proof::prepare_base_origin(layout, &cohort[index], limits).await?, + proof::prepare_base_origin(layout, cohort[index], limits).await?, )) }) .buffer_unordered(READ_CONCURRENCY); @@ -43,7 +43,7 @@ pub(super) async fn verify( // All bounded operations have joined/dropped before a larger graph // takes the original serial working set. No partial plan grants CAS. for index in serial { - proof::verify_base(layout, &cohort[index], limits).await?; + proof::verify_base(layout, cohort[index], limits).await?; } } Ok(()) diff --git a/crates/cellule-runtime/src/node/bundle/verification/mod.rs b/crates/cellule-runtime/src/node/bundle/verification/mod.rs index 30ed2ee5..dc60082e 100644 --- a/crates/cellule-runtime/src/node/bundle/verification/mod.rs +++ b/crates/cellule-runtime/src/node/bundle/verification/mod.rs @@ -7,6 +7,7 @@ use tokio::sync::Semaphore; const READ_CONCURRENCY: usize = 8; mod base; +pub(super) use base::verify as verify_bases; // Together scratch and metadata replace the released fresh-origin buffer; // they never add to the producer's original 20-MiB working reservation. const SCRATCH_BYTES: u64 = MAX_BUNDLE_BYTES / 2; @@ -145,7 +146,10 @@ pub(super) async fn verify_cohort( } // Base and historical verification occupy the released origin allowance // in disjoint phases. No historical facts or windows coexist with bases. - base::verify(layout, bindings, limits).await?; + { + let bases = bindings.iter().collect::>(); + verify_bases(layout, &bases, limits).await?; + } let windows = windows(bindings)?; let facts = Mutex::new( bindings From 41bc99e8324c0280da4ca3710374f13d6c7ea934 Mon Sep 17 00:00:00 2001 From: forhappy Date: Fri, 9 Oct 2026 16:04:45 -0700 Subject: [PATCH 094/102] Revert "Verify checkpoint bases in the admitted publication cohort" This reverts commit 634bd90e0a3415301f02b922d65e8e4c52a6db55. --- .../docs/write-performance-design.md | 9 +- .../src/node/bundle/closure.rs | 28 +-- .../bundle/tests/index/checkpoint_cohort.rs | 181 ------------------ .../src/node/bundle/tests/index/mod.rs | 1 - .../src/node/bundle/verification/base.rs | 8 +- .../src/node/bundle/verification/mod.rs | 6 +- 6 files changed, 10 insertions(+), 223 deletions(-) delete mode 100644 crates/cellule-runtime/src/node/bundle/tests/index/checkpoint_cohort.rs diff --git a/crates/cellule-runtime/docs/write-performance-design.md b/crates/cellule-runtime/docs/write-performance-design.md index f2a45d45..ad69ed25 100644 --- a/crates/cellule-runtime/docs/write-performance-design.md +++ b/crates/cellule-runtime/docs/write-performance-design.md @@ -146,14 +146,7 @@ Startup admission precedes installation of the irreversible feed. Selection and exact root checkpoints use the same original binding/heartbeat authority; 512 checkpoint requests are bounded and their callbacks join before Cell closure. Fair turns alternate queued native work and -checkpoint cohorts. Exact checkpoints reuse selection's canonical fresh-base -verifier for changed participants: at most eight small-root operations with a -4-MiB operation charge, within the original producer working reservation. -Larger graphs stay serial. No catalog upload or CAS occurs before every original -participant joins verification; no prior availability proof is cached. The real -held-root/body fixture verifies this overlap, cancellation and exact cold results; -application throughput still requires a fresh measurement. -The Fleet SQL example installs this producer; the current +checkpoint cohorts. The Fleet SQL example installs this producer; the current Bucket-only performance adapter bypasses it. Each selection reads its complete new cohort object once from origin, compares diff --git a/crates/cellule-runtime/src/node/bundle/closure.rs b/crates/cellule-runtime/src/node/bundle/closure.rs index 0c9e82ad..2fcfb408 100644 --- a/crates/cellule-runtime/src/node/bundle/closure.rs +++ b/crates/cellule-runtime/src/node/bundle/closure.rs @@ -25,8 +25,6 @@ impl NodeDirectory { /// index upload and one shared node CAS. Each root must match its opaque /// complete-capture proof; any missing/stale participant leaves the catalog /// unchanged. Callers retain their admissions until this operation joins. - /// Fresh small-root verification overlaps at most eight operations within - /// the original 4-MiB working allowance; larger graphs remain serial. pub async fn checkpoint_bundle_cells( &self, observed: &VersionedNodeAdvertisement, @@ -59,7 +57,7 @@ impl NodeDirectory { let mut catalog = store::load_catalog_cells(&self.layout, observed.advertisement.session, head, &cells) .await?; - let mut changed = Vec::with_capacity(checkpoints.len()); + let mut changed = false; for (authority, proof) in checkpoints { let pin = proof.binding(); let binding = catalog.binding_mut(pin.digest)?; @@ -92,34 +90,16 @@ impl NodeDirectory { continue; } binding.control = control.clone(); + verify_base(&self.layout, binding, limits).await?; binding.locators.drain(..prefix); if binding.locators.is_empty() { binding.first_commit = binding.selected_commit; } - changed.push(pin.digest); + changed = true; } - if changed.is_empty() { + if !changed { return Ok(observed.clone()); } - { - // These are only the original changed participants, never siblings. - // Catalog edits remain private until every fresh root/dependency - // joins the same admitted verifier used by shared selection. - let bases = catalog - .bindings - .iter() - .filter(|binding| { - binding - .control - .bundle_binding - .is_some_and(|pin| changed.contains(&pin.digest)) - }) - .collect::>(); - if bases.len() != changed.len() { - return Err(Error::Node("bundle checkpoint participant is absent")); - } - verification::verify_bases(&self.layout, &bases, limits).await?; - } let prepared = self.upload_catalog(Some(head), catalog, &[]).await?; self.select_catalog(observed, &prepared, now_ms).await } diff --git a/crates/cellule-runtime/src/node/bundle/tests/index/checkpoint_cohort.rs b/crates/cellule-runtime/src/node/bundle/tests/index/checkpoint_cohort.rs deleted file mode 100644 index e449820d..00000000 --- a/crates/cellule-runtime/src/node/bundle/tests/index/checkpoint_cohort.rs +++ /dev/null @@ -1,181 +0,0 @@ -use super::*; -use std::sync::atomic::Ordering; - -#[tokio::test] -async fn checkpoint_fresh_roots_overlap_within_original_working_admission() { - held_checkpoint_bases(12).await; -} - -#[tokio::test] -async fn checkpoint_fresh_bodies_join_before_upload_and_exact_cold_recovery() { - held_checkpoint_bases(13).await; -} - -async fn held_checkpoint_bases(mode: u8) { - let faults = Arc::new(super::super::faults::ReplyFault::default()); - let mut f = Fixture::with_store(faults.clone()).await; - let mut cells = Vec::new(); - for number in 4..14 { - cells.push(f.cell(number).await); - f.heartbeat().await; - } - let mut frames = Vec::new(); - let mut assigned = Vec::new(); - for cell in &mut cells { - let (_, capture, range) = f.append(cell, 2); - frames.extend(capture); - assigned.push(range); - } - let now = f.node.advertisement().issued_at_ms(); - let proposal = f - .directory - .prepare_node_bundle(&f.node, &frames, &assigned, now) - .await - .unwrap(); - let (node, proofs) = f - .directory - .select_node_bundle(&f.node, &proposal, &f.lease, Limits::default(), now) - .await - .unwrap(); - f.node = node; - for proof in &proofs { - let cell = cells - .iter() - .find(|cell| cell.control.value().bundle_binding == Some(proof.binding())) - .unwrap(); - f.publisher(cell).materialize_bundle(proof).await.unwrap(); - f.heartbeat().await; - } - let checkpoints: Vec<_> = proofs - .iter() - .map(|proof| { - let cell = cells - .iter() - .find(|cell| cell.control.value().bundle_binding == Some(proof.binding())) - .unwrap(); - (&cell.authority, proof) - }) - .collect(); - f.count.reset(); - faults.mode.store(mode, Ordering::SeqCst); - let now = f.node.advertisement().issued_at_ms(); - { - let checkpoint = - f.directory - .checkpoint_bundle_cells(&f.node, &checkpoints, Limits::default(), now); - tokio::pin!(checkpoint); - let overlap = async { - loop { - let started = faults.node_started.notified(); - if faults.base_started.load(Ordering::SeqCst) >= 8 { - break; - } - started.await; - } - }; - tokio::select! { - _ = &mut checkpoint => panic!("held checkpoint bases must prevent upload and CAS"), - result = tokio::time::timeout(std::time::Duration::from_secs(5), overlap) => { - println!("checkpoint original dependencies started before release={}", faults.base_started.load(Ordering::SeqCst)); - result.unwrap(); - }, - } - assert_eq!( - faults.base_started.load(Ordering::SeqCst), - 8, - "checkpoint verification must overlap only the admitted first eight bases" - ); - } - assert_eq!(f.count.put_requests(), 0); - faults.mode.store(0, Ordering::SeqCst); - faults.node_resume.notify_waiters(); - let original = cells[0] - .authority - .load(cells[0].control.value().cell) - .await - .unwrap() - .unwrap() - .value() - .ltx_root() - .unwrap(); - let missing = f.layout.incarnation_object_path( - &original.cell, - &original.incarnation, - &original.digest, - cellule_ltx::CellObjectKind::Root, - ); - f.count.block_body_reads_for(&missing); - f.count.reset(); - assert!( - f.directory - .checkpoint_bundle_cells(&f.node, &checkpoints, Limits::default(), now) - .await - .is_err() - ); - assert_eq!(f.count.put_requests(), 0); - f.count.unblock_body_reads_for(&missing); - f.count.reset(); - f.node = f - .directory - .checkpoint_bundle_cells(&f.node, &checkpoints, Limits::default(), now) - .await - .unwrap(); - assert_eq!( - f.count - .requests() - .iter() - .filter(|request| request.location.ends_with(".root")) - .count(), - cells.len(), - "each eligible checkpoint must freshly authenticate its original root once" - ); - assert_eq!(f.count.put_requests(), 2); - for (number, cell) in cells.iter().enumerate() { - let current = cell - .authority - .load(cell.control.value().cell) - .await - .unwrap() - .unwrap(); - let selected = f - .directory - .load_bundle_coverage(&cell.authority, ¤t, Limits::default()) - .await - .unwrap(); - assert_eq!(selected.locator_count(), 0); - assert_eq!(selected.commit_sequence(), 2); - let root = selected.base().unwrap(); - let cold = CellReplica::new( - f.layout.clone(), - root.cell, - root.incarnation, - Limits::default(), - ) - .unwrap(); - let destination = f - .scratch - .path() - .join(format!("checkpoint-cohort-{number}.sqlite")); - cold.open_root(&root) - .await - .unwrap() - .restore(&destination) - .await - .unwrap(); - let db = rusqlite::Connection::open(destination).unwrap(); - let outcomes: Vec<(String, String)> = db - .prepare("SELECT request, result FROM outcomes ORDER BY request") - .unwrap() - .query_map([], |row| Ok((row.get(0)?, row.get(1)?))) - .unwrap() - .collect::>() - .unwrap(); - assert_eq!( - outcomes, - vec![ - ("request-2".into(), "result-2".into()), - ("seed".into(), "original".into()) - ] - ); - } -} diff --git a/crates/cellule-runtime/src/node/bundle/tests/index/mod.rs b/crates/cellule-runtime/src/node/bundle/tests/index/mod.rs index 4ca166da..158abc16 100644 --- a/crates/cellule-runtime/src/node/bundle/tests/index/mod.rs +++ b/crates/cellule-runtime/src/node/bundle/tests/index/mod.rs @@ -41,7 +41,6 @@ fn head_session(f: &Fixture) -> SessionId { mod base_cohort; mod bootstrap; mod checkpoint; -mod checkpoint_cohort; mod cohort; mod compatibility; mod copy_on_write; diff --git a/crates/cellule-runtime/src/node/bundle/verification/base.rs b/crates/cellule-runtime/src/node/bundle/verification/base.rs index a9db0d62..52a3595c 100644 --- a/crates/cellule-runtime/src/node/bundle/verification/base.rs +++ b/crates/cellule-runtime/src/node/bundle/verification/base.rs @@ -1,9 +1,9 @@ //! Admitted fresh small-base verification; larger graphs stay serial. use super::*; -pub(in crate::node::bundle) async fn verify( +pub(super) async fn verify( layout: &cellule_ltx::CellStorageLayout, - bindings: &[&Binding], + bindings: &[Binding], limits: cellule_ltx::Limits, ) -> Result<()> { for cohort in bindings.chunks(READ_CONCURRENCY) { @@ -13,7 +13,7 @@ pub(in crate::node::bundle) async fn verify( .map(|index| async move { Ok::<_, Error>(( index, - proof::prepare_base_origin(layout, cohort[index], limits).await?, + proof::prepare_base_origin(layout, &cohort[index], limits).await?, )) }) .buffer_unordered(READ_CONCURRENCY); @@ -43,7 +43,7 @@ pub(in crate::node::bundle) async fn verify( // All bounded operations have joined/dropped before a larger graph // takes the original serial working set. No partial plan grants CAS. for index in serial { - proof::verify_base(layout, cohort[index], limits).await?; + proof::verify_base(layout, &cohort[index], limits).await?; } } Ok(()) diff --git a/crates/cellule-runtime/src/node/bundle/verification/mod.rs b/crates/cellule-runtime/src/node/bundle/verification/mod.rs index dc60082e..30ed2ee5 100644 --- a/crates/cellule-runtime/src/node/bundle/verification/mod.rs +++ b/crates/cellule-runtime/src/node/bundle/verification/mod.rs @@ -7,7 +7,6 @@ use tokio::sync::Semaphore; const READ_CONCURRENCY: usize = 8; mod base; -pub(super) use base::verify as verify_bases; // Together scratch and metadata replace the released fresh-origin buffer; // they never add to the producer's original 20-MiB working reservation. const SCRATCH_BYTES: u64 = MAX_BUNDLE_BYTES / 2; @@ -146,10 +145,7 @@ pub(super) async fn verify_cohort( } // Base and historical verification occupy the released origin allowance // in disjoint phases. No historical facts or windows coexist with bases. - { - let bases = bindings.iter().collect::>(); - verify_bases(layout, &bases, limits).await?; - } + base::verify(layout, bindings, limits).await?; let windows = windows(bindings)?; let facts = Mutex::new( bindings From 9f7738a94770910f896d1d61bd3bed2abcda6372 Mon Sep 17 00:00:00 2001 From: forhappy Date: Fri, 9 Oct 2026 16:07:31 -0700 Subject: [PATCH 095/102] Record rejected checkpoint cohort and fresh performance evidence --- .../docs/write-performance-design.md | 14 ++- docs/pr67-checkpoint-cohort-measurement.md | 118 ++++++++++++++++++ 2 files changed, 131 insertions(+), 1 deletion(-) create mode 100644 docs/pr67-checkpoint-cohort-measurement.md diff --git a/crates/cellule-runtime/docs/write-performance-design.md b/crates/cellule-runtime/docs/write-performance-design.md index ad69ed25..9127df86 100644 --- a/crates/cellule-runtime/docs/write-performance-design.md +++ b/crates/cellule-runtime/docs/write-performance-design.md @@ -1,6 +1,18 @@ # Node write and read performance design -The latest [sparse-shard diagnostic](../../../docs/pr67-shard-window-measurement.md) +The latest [checkpoint-cohort diagnostic](../../../docs/pr67-checkpoint-cohort-measurement.md) +rejects and reverts bounded parallel checkpoint verification. The controlled +fixture overlaps eight fresh roots rather than one, but application TPS falls +583.20→537.28/s and successful scheduled p99 barely changes, 368.89→367.03 ms. +Fresh celld completes 1,999.90/s with 14.48-ms p99. The prototype passes all +62,011 warm/cold ACK checks; the retained baseline has 18 post-SQL publication +capacity fences, 1,652 warm HTTP 503s and an owner cleanup timeout. Shared root +packing counters are zero in both arms: bundle materialization bypasses the +existing shared producer. Prioritize that integration, publication bytes per +write and the remaining capacity-fencing failure. All profiles remain unqualified; +the acceptance gates below are unchanged. + +The preceding [sparse-shard diagnostic](../../../docs/pr67-shard-window-measurement.md) rejects and reverts wide metadata reads. Its component fixture improves 49→2 reads, but fresh application TPS falls 510.13→300.63/s and successful scheduled p99 rises 1,314.72→2,554.88 ms. Node-authority range bytes rise 61.17% despite diff --git a/docs/pr67-checkpoint-cohort-measurement.md b/docs/pr67-checkpoint-cohort-measurement.md new file mode 100644 index 00000000..fa06741f --- /dev/null +++ b/docs/pr67-checkpoint-cohort-measurement.md @@ -0,0 +1,118 @@ +# PR 67: rejected checkpoint verification cohort + +**The checkpoint prototype is reverted. PR #67 remains a draft; performance +parity is unmet.** A fresh Docker comparison completes 583.20 Fleet writes/s +for retained Cellule, 537.28/s for the prototype and 1,999.90/s for celld. +Successful scheduled p99 is 368.89, 367.03 and 14.48 ms respectively. The +component change establishes no acceptable application performance gain. + +## Experiment and disposition + +Prototype `634bd90e0a3415301f02b922d65e8e4c52a6db55` routes changed checkpoint +participants through the existing bounded base verifier used by bundle +selection. Up to eight fresh small-root operations overlap inside its original +4-MiB phase allowance and 20-MiB producer working budget. Larger roots keep +the serial fallback. Only the original changed participants are verified; +every required root and dependency must finish before catalog upload or CAS. +There is no across-call availability cache or persisted-format change. + +The unchanged five-second controlled assertion fails on the baseline with +only one original dependency started. Both fixtures pass three repetitions +on the prototype with exactly eight overlapping dependencies, no ninth read, +no PUT after cancellation or missing required root bodies, and exact cold +mutation/retry recovery for all ten Cells per fixture. All 101 bundle tests pass. + +Revert `41bc99e` restores production byte-identically to the preceding PR head +`6450734`. The original admission-before-ordering, eight-round follower pipeline, +follower group commit, fresh-origin encoder reuse and command-credit handoff +remain. The rejected source, fixtures and measurements stay in external evidence. + +## Fresh application results + +Same SQL ledger workload: 2,000 uniformly active Cells, 96-byte values, one +owner and two followers, WAL NORMAL/tmpfs, 128 clients and queue slots, +2,000 offered writes/s, 30-second warmup and 60-second measured window. +Retained runtime `cf4785c` and prototype `634bd90` have identical driver/auditor +binaries, fixtures, workload and pinned images. Fresh celld is +`f2bf648663a610eefde71f3547ad61e9b896b1f0`. Builds, tests and independent replay +do not overlap timed windows; all original controllers join before analysis. + +| Arm | Successful writes/s | Successful scheduled p99 ms | Errors | Dropped offers | +| --- | ---: | ---: | ---: | ---: | +| Retained Cellule | 583.20 | 368.89 | 49,119 | 35,889 | +| Rejected checkpoint prototype | 537.28 | 367.03 | 52,127 | 35,522 | +| Fresh celld | 1,999.90 | 14.48 | 0 | 0 | + +TPS counts completions inside the measured window. Scheduled p99 includes +measured successful offers through client drain and excludes failures. Successful +request p99 is 190.05, 217.50 and 13.36 ms respectively. Prototype TPS falls +7.87%; request p99 rises 14.44%. Its 114 trailing successes do not count toward +TPS; celld has six. Every arm has measured successes on all 2,000 Cells. +One short sequential pair establishes no repeatable causal attribution. + +Independent journal replay reconciles all original outputs, payload bytes, +per-Cell counts, offers, attempts, trailing completions and complete ACK provenance. +Retained Cellule has 49,101 `unavailable` and 18 `outcome_unknown` measured errors; +all prototype errors are `unavailable`. Warmup errors/drops are 24,425/7,963, +24,447/7,894 and 0/269 respectively. Celld's warmup drops remain a failed gate. + +| Arm | Complete ACKs | Warm mutation / original retry audit | Cold audit / fleet drain | +| --- | ---: | --- | --- | +| Retained Cellule | 64,605 | 1,652 HTTP 503s; 62,953 retries checked | Cold not reached; owner cleanup exceeds 120 s | +| Rejected prototype | 62,011 | All pass | All pass; 46.73 s | +| Fresh celld | 181,732 | All pass | All pass; 13.31 s | + +Retained Cellule logs 18 owner fences with `Capacity("pending publication bytes")`; +active Cells fall 2,000→1,982. The prototype logs no owner fences and drains all +2,000 Cells to Idle. Audit unavailability and unknown outcomes do not establish +lost acknowledged data, but the baseline fails availability and draining gates. +The prototype's passed ACK cohort does not qualify the complete failure matrix. + +## Remaining publication work + +| Measured-window metric | Retained Cellule | Rejected prototype | +| --- | ---: | ---: | +| Node-authority range starts / bytes | 121,161 / 1,389,532,747 | 111,387 / 1,271,899,521 | +| Range bytes / successful in-window write | 39,710 | 39,455 | +| Immutable GET starts / bytes | 113,879 / 466,391,362 | 104,518 / 406,223,379 | +| Mean ordered-lock wait ms | 0.00034 | 0.00074 | +| Mean native submission ms | 0.13844 | 0.17528 | +| Separate mean Fleet-proof wait ms | 59.97 | 58.96 | +| Follower frames / sync, two members | 10.95 / 10.94 | 11.18 / 11.18 | +| Unpublished native bytes, start→end | 38,463,948→54,826,539 | 37,329,092→47,068,226 | + +All seven native phase counts and nanosecond totals reconcile exactly. Fleet +proof waits are a separate overlapping cohort, not a serial throughput partition. +Publication bytes per successful write barely change. Retained runtime memory +grows 49,544,295→57,478,675 bytes; prototype memory grows 49,494,363→62,988,087. +Neither establishes stable debt at the offered load. + +Both arms record zero shared root-packing cohorts. `materialize_bundle_with_due` +calls `prepare_recovered_overlay` directly, bypassing the existing bounded shared +publication producer. The next experiment should connect eligible exact recovered +overlays to that canonical producer, preserving lineage, proofs, byte/descriptor +admission, lifetime ownership, collection and joined shutdown. Measure total +publication requests and bytes per successful write, not only read concurrency. +Also reproduce and fix post-SQL publication capacity fencing without raising +limits; the fresh baseline shows the prior handoff has not eliminated that failure. + +## Verification and limits + +The prototype passes all 13 contributor routes in an isolated snapshot: 1,995 +reported workspace test/doctest executions including child-process reporting, +38 ignored environment tests, 60 local LTX tests, Rust 1.97/1.99 Clippy, targets, +rustdoc and boundary/layout/document/contract/script gates. The invalid initial +cache-overlap compilation is preserved and is not passing evidence; the separate +controlled baseline reaches and fails both original assertions. Final document +gates cover the report/design edits after the byte-identical production restore. + +All canonical qualification reports remain false. The shared Docker VM has eight +CPUs and 8,306,286,592 memory bytes across all roles; it does not qualify a +dedicated 8-vCPU/16-GiB owner. tmpfs does not qualify physical-media durability. +Cellule uses pinned mTLS/signed protobuf and celld internal HTTP on loopback; +no transport-cost attribution is claimed. No new Bucket, read-only or mixed +result is claimed. Three matched repetitions of at least five minutes and the +unchanged latency, zero-error/drop, complete recovery, read and debt gates remain. + +Raw evidence stays outside Git at +`/Volumes/Workspace/crabbuild-target/native-checkpoint-cohort-20261009`. From c2c0ef10b08b9844401ca9e11397e9321a744f91 Mon Sep 17 00:00:00 2001 From: forhappy Date: Fri, 9 Oct 2026 16:38:47 -0700 Subject: [PATCH 096/102] Route recovered overlays through bounded shared publication --- crates/cellule-ltx/docs/publication.md | 13 ++ .../cellule-ltx/src/environment/host/mod.rs | 18 ++ .../cellule-ltx/src/environment/host/tests.rs | 69 +++++++ .../cellule-ltx/src/replica/bundle_inputs.rs | 109 +++++++++++ crates/cellule-ltx/src/replica/mod.rs | 15 ++ crates/cellule-ltx/src/replica/prepare.rs | 109 ++--------- crates/cellule-ltx/src/replica/shared/mod.rs | 70 ++++++- crates/cellule-ltx/tests/cell/roots.rs | 1 + .../cellule-ltx/tests/cell/roots/lifecycle.rs | 7 + .../tests/cell/roots/shared_recovery.rs | 176 +++++++++++++++++ .../docs/write-performance-design.md | 9 + .../bundle/tests/materialization_shared.rs | 180 ++++++++++++++++++ .../src/node/bundle/tests/mod.rs | 1 + crates/cellule-runtime/src/publication/mod.rs | 10 +- .../src/publication/recovered.rs | 88 +++++++++ .../src/publication/shared/mod.rs | 83 ++++++-- .../src/publication/shared/tests.rs | 4 +- 17 files changed, 850 insertions(+), 112 deletions(-) create mode 100644 crates/cellule-ltx/src/environment/host/tests.rs create mode 100644 crates/cellule-ltx/src/replica/bundle_inputs.rs create mode 100644 crates/cellule-ltx/tests/cell/roots/shared_recovery.rs create mode 100644 crates/cellule-runtime/src/node/bundle/tests/materialization_shared.rs create mode 100644 crates/cellule-runtime/src/publication/recovered.rs diff --git a/crates/cellule-ltx/docs/publication.md b/crates/cellule-ltx/docs/publication.md index 32668a53..3b7898f9 100644 --- a/crates/cellule-ltx/docs/publication.md +++ b/crates/cellule-ltx/docs/publication.md @@ -23,6 +23,9 @@ sequenceDiagram | `CellReplica::prepare` | Verifies cuts and writes immutable root dependencies. | | `prepare_bundle` | Selects this Cell's exact rows from a shared bundle. | | `prepare_recovered_overlay` | Verifies the exact predecessor and final position; small independent tails use the canonical coalescer and native pack. | +| `shared_recovered_captures` | Verifies eligible overlay rows through the same input reader; original chain facts survive coalescing for the canonical shared root factory. | +| `RecoveryOverlay::shared_input_upper_bound` | Provides conservative original row/body/index bounds before allocation; grants no verification or authority. | +| `with_preparation_resource` | Keeps a caller's existing memory/artifact admission alive in dispatched jobs after waiter cancellation; reserves nothing and grants no authority. | | `prepare_compaction` | Rewrites representation without changing logical state. | | `prepare_after_compaction` | Appends to a private compaction while retaining its original authority predecessor. | | `try_admit_scheduled_compaction` | Returns a scoped clone with existing dirty/recovery admission, or defers without waiting behind a queued cohort. | @@ -88,6 +91,16 @@ individual row, changed image or output retains the ordinary bundle path. `prepare_bundle` preserves shared bundle references. No root format, authority rule or host resource ceiling changes. +Selected bundle materialization can feed those same verified small inputs into +the runtime's existing shared publication producer. The original 256-KiB object +and 64-row bounds apply before reduction; larger tails retain ordinary recovery +preparation. Retained-memory pressure selects that direct fallback immediately, +without holding a dirty permit while waiting for the shared upload. Each Cell +still verifies its fresh predecessor, complete original chain and exact endpoint, +retains native lineage, then selects its own root under the existing fenced CAS. +Shared-object origin verification remains complete; lower upload counts alone +do not establish lower total publication bytes or application TPS. + A representation-only compaction can remain private while its successor append uploads. `prepare_after_compaction` verifies that the compaction preserves the predecessor's position, commit sequence, Cell and incarnation. The runtime selects diff --git a/crates/cellule-ltx/src/environment/host/mod.rs b/crates/cellule-ltx/src/environment/host/mod.rs index 1e22d298..7b7aad1a 100644 --- a/crates/cellule-ltx/src/environment/host/mod.rs +++ b/crates/cellule-ltx/src/environment/host/mod.rs @@ -14,6 +14,9 @@ use std::{ mod admission; mod budget; +#[cfg(all(test, feature = "replica"))] +mod tests; + pub use budget::{DiskBudget, DiskReservation}; #[cfg(feature = "replica")] @@ -200,6 +203,8 @@ pub struct Host { dirty: Option>, #[cfg(feature = "replica")] scratch: Option>, + #[cfg(feature = "replica")] + preparation_resource: Option>, } #[cfg(feature = "replica")] @@ -724,6 +729,15 @@ impl Host { .map_err(crate::LtxError::Io) } + #[cfg(feature = "replica")] + pub(crate) fn with_preparation_resource( + mut self, + resource: Arc, + ) -> Self { + self.preparation_resource = Some(resource); + self + } + #[cfg(feature = "replica")] pub(crate) async fn run( &self, @@ -749,6 +763,7 @@ impl Host { let recovery = self.recovery.clone(); let dirty = self.dirty.clone(); let scratch = self.scratch.clone(); + let preparation_resource = self.preparation_resource.clone(); self.executor.dispatch(Box::new(move || { // Dispatched work can outlive its future. Keep admission with the // job, not the waiter, so cancellation cannot oversubscribe the pool. @@ -760,6 +775,7 @@ impl Host { drop(recovery); drop(dirty); drop(scratch); + drop(preparation_resource); drop(resource); drop(permit); let _ = send.send(result); @@ -856,6 +872,8 @@ impl Default for Host { dirty: None, #[cfg(feature = "replica")] scratch: None, + #[cfg(feature = "replica")] + preparation_resource: None, } } } diff --git a/crates/cellule-ltx/src/environment/host/tests.rs b/crates/cellule-ltx/src/environment/host/tests.rs new file mode 100644 index 00000000..1e9b913f --- /dev/null +++ b/crates/cellule-ltx/src/environment/host/tests.rs @@ -0,0 +1,69 @@ +use super::*; +use crate::environment::executor::Worker; + +#[derive(Default)] +struct HeldExecutor { + jobs: Mutex>>, + started: tokio::sync::Notify, +} + +impl Executor for HeldExecutor { + fn dispatch(&self, job: Box) -> io::Result<()> { + self.jobs.lock().unwrap().push(job); + self.started.notify_one(); + Ok(()) + } + + fn start_worker(&self, job: Box) -> io::Result> { + TokioExecutor.start_worker(job) + } +} + +struct PreparationResource(Arc); + +impl HostResourcePermit for PreparationResource {} + +impl Drop for PreparationResource { + fn drop(&mut self) { + self.0.store(true, Ordering::SeqCst); + } +} + +#[tokio::test] +async fn preparation_resources_survive_cancelled_native_waiters_and_release_on_completion() { + for mode in 0..3 { + let executor = Arc::new(HeldExecutor::default()); + let jobs = Arc::new(tokio::sync::Semaphore::new(1)); + let released = Arc::new(AtomicBool::new(false)); + let host = Host::default() + .with_executor(executor.clone()) + .with_job_slots(jobs.clone()) + .with_preparation_resource(Arc::new(PreparationResource(released.clone()))); + let mut work = Box::pin(async move { + host.run(move || { + assert_ne!(mode, 2, "injected native failure"); + 7 + }) + .await + }); + tokio::time::timeout(Duration::from_secs(1), async { + tokio::select! { + result = &mut work => panic!("held original job completed: {result:?}"), + _ = executor.started.notified() => {} + } + }) + .await + .unwrap(); + drop(work); + assert!(!released.load(Ordering::SeqCst)); + assert_eq!(jobs.available_permits(), 0); + let job = executor.jobs.lock().unwrap().pop().unwrap(); + if mode == 1 { + drop(job); + } else { + job(); + } + assert!(released.load(Ordering::SeqCst)); + assert_eq!(jobs.available_permits(), 1); + } +} diff --git a/crates/cellule-ltx/src/replica/bundle_inputs.rs b/crates/cellule-ltx/src/replica/bundle_inputs.rs new file mode 100644 index 00000000..aa907f5b --- /dev/null +++ b/crates/cellule-ltx/src/replica/bundle_inputs.rs @@ -0,0 +1,109 @@ +//! Canonical verified Cell inputs before bundle representation reduction. +use super::*; + +pub(super) struct BundleInputs { + pub inputs: Vec, + pub target: Position, + pub independent: bool, +} + +impl CellReplica { + pub(super) fn validate_recovery_overlay(&self, overlay: &RecoveryOverlay) -> Result<()> { + if overlay.predecessor.cell != self.cell + || overlay.predecessor.incarnation != self.incarnation + || overlay.final_commit_sequence <= overlay.predecessor.commit_sequence + { + return Err(LtxError::InvalidState("recovery overlay scope")); + } + let (repository, epoch) = crate::bundle::cell_identity(&self.cell, &self.incarnation); + let final_position = overlay + .bundle + .rows() + .iter() + .rfind(|row| row.repository == repository && row.epoch == epoch) + .map(|row| row.info.position()) + .ok_or(LtxError::TxNotAvailable)?; + if final_position != overlay.final_position { + return Err(LtxError::ChecksumMismatch); + } + Ok(()) + } + + pub(super) fn read_bundle_inputs( + &self, + bundle: &crate::bundle::Bundle, + mut independent: bool, + ) -> Result { + if bundle.len() > self.limits.max_plan_bytes { + return Err(LtxError::Limit(crate::LimitKind::CellBundleBytes)); + } + let (repository, epoch) = crate::bundle::cell_identity(&self.cell, &self.incarnation); + let bundle_digest = bundle.digest(); + let mut inputs: Vec = Vec::new(); + let mut selected_bytes = 0_u64; + let mut independent_bytes = 0_u64; + for (index, row) in bundle.rows().iter().enumerate() { + if row.repository != repository || row.epoch != epoch { + continue; + } + selected_bytes = selected_bytes + .checked_add(row.info.size_bytes) + .ok_or(LtxError::Limit(crate::LimitKind::CapturedCellBundleBytes))?; + if row.info.size_bytes > self.limits.max_file_bytes + || selected_bytes > self.limits.max_plan_bytes + { + return Err(LtxError::Limit(crate::LimitKind::CapturedCellBundleBytes)); + } + let bytes = bundle.read_segment(index)?; + let (file, size, digest, pages) = crate::ltx::inspect_bytes_with_index(&bytes)?; + if size != row.info.size_bytes + || digest != row.info.blake3 + || crate::SegmentInfo::from_inspected(&file, size, digest) != row.info + { + return Err(LtxError::ChecksumMismatch); + } + self.admit_segment_representation(&row.info, pages.len() * crate::paged::ENTRY_BYTES)?; + let index_bytes = Bytes::from(crate::paged::encode_index_from_pages(&pages)?); + // Release earlier frozen rows as soon as the original small-pack + // allowance fails. A shared producer cannot widen this bound. + if independent { + match independent_bytes + .checked_add(packed::HEADER_BYTES) + .and_then(|size| size.checked_add(row.info.size_bytes)) + .and_then(|size| size.checked_add(index_bytes.len() as u64)) + .filter(|size| *size <= upload::SINGLE_PUT_BYTES) + { + Some(size) => independent_bytes = size, + None => { + independent = false; + for input in &mut inputs { + input.body = AppendBody::Bundle; + } + } + } + } + inputs.push(AppendInput { + info: row.info.clone(), + location: BodyLocation::Bundle { + digest: bundle_digest, + offset: row.offset, + }, + index: index_bytes, + body: if independent { + AppendBody::Frozen(bytes) + } else { + AppendBody::Bundle + }, + }); + } + let target = inputs + .last() + .map(|input| input.info.position()) + .ok_or(LtxError::TxNotAvailable)?; + Ok(BundleInputs { + inputs, + target, + independent, + }) + } +} diff --git a/crates/cellule-ltx/src/replica/mod.rs b/crates/cellule-ltx/src/replica/mod.rs index 384fa32e..d9d5f984 100644 --- a/crates/cellule-ltx/src/replica/mod.rs +++ b/crates/cellule-ltx/src/replica/mod.rs @@ -13,6 +13,7 @@ use futures_util::{StreamExt as _, TryStreamExt as _, stream}; use crate::{CaptureBatch, Host, Limits, LtxError, Position, Result}; +mod bundle_inputs; mod cache; mod coalesce; mod compaction; @@ -822,6 +823,20 @@ impl CellReplica { self } + /// Retains a caller's pre-admitted preparation resources through native jobs. + /// + /// This reserves no resources and grants no root or writer authority. Use a + /// scoped clone: dispatched jobs keep the token after waiter cancellation, + /// and shared inputs keep it until their preparation owner releases them. + #[must_use] + pub fn with_preparation_resource( + mut self, + resource: Arc, + ) -> Self { + self.host = self.host.with_preparation_resource(resource); + self + } + /// Exclusively creates a fresh local database using this replica's host and limits. /// /// The destination and SQLite sidecars must not exist. A failed open leaves diff --git a/crates/cellule-ltx/src/replica/prepare.rs b/crates/cellule-ltx/src/replica/prepare.rs index 177e2115..01ddf119 100644 --- a/crates/cellule-ltx/src/replica/prepare.rs +++ b/crates/cellule-ltx/src/replica/prepare.rs @@ -236,23 +236,7 @@ impl CellReplica { overlay: &RecoveryOverlay, schema: u32, ) -> Result { - if overlay.predecessor.cell != self.cell - || overlay.predecessor.incarnation != self.incarnation - || overlay.final_commit_sequence <= overlay.predecessor.commit_sequence - { - return Err(LtxError::InvalidState("recovery overlay scope")); - } - let (repository, epoch) = crate::bundle::cell_identity(&self.cell, &self.incarnation); - let final_position = overlay - .bundle - .rows() - .iter() - .rfind(|row| row.repository == repository && row.epoch == epoch) - .map(|row| row.info.position()) - .ok_or(LtxError::TxNotAvailable)?; - if final_position != overlay.final_position { - return Err(LtxError::ChecksumMismatch); - } + self.validate_recovery_overlay(overlay)?; let mut replica = self.clone(); replica.host = self.host.for_dirty().await?; let prepared = replica @@ -289,81 +273,30 @@ impl CellReplica { self.validate_append_sequence(&base_graph, commit_sequence)?; let (repository, epoch) = crate::bundle::cell_identity(&self.cell, &self.incarnation); - let bundle_digest = bundle.digest(); - let mut inputs: Vec = Vec::new(); - let mut selected_bytes = 0_u64; - let mut independent = usage == BundleUse::IndependentRecovery; - let mut independent_bytes = 0_u64; + let bundle_inputs = + self.read_bundle_inputs(bundle, usage == BundleUse::IndependentRecovery)?; + let mut inputs = bundle_inputs.inputs; + let target = bundle_inputs.target; + let mut independent = bundle_inputs.independent; let mut prospective = base_graph .as_ref() .map(|graph| graph.descriptors.clone()) .unwrap_or_default(); - for (index, row) in bundle.rows().iter().enumerate() { - if row.repository != repository || row.epoch != epoch { - continue; - } - selected_bytes = selected_bytes - .checked_add(row.info.size_bytes) - .ok_or(LtxError::Limit(crate::LimitKind::CapturedCellBundleBytes))?; - if row.info.size_bytes > self.limits.max_file_bytes - || selected_bytes > self.limits.max_plan_bytes - { - return Err(LtxError::Limit(crate::LimitKind::CapturedCellBundleBytes)); - } - prospective.push(SegmentDescriptor::bundled( - row.info.clone(), - [0; 32], - 0, - bundle_digest, - row.offset, - )); - let bytes = bundle.read_segment(index)?; - let (file, size, digest, pages) = crate::ltx::inspect_bytes_with_index(&bytes)?; - if size != row.info.size_bytes - || digest != row.info.blake3 - || crate::SegmentInfo::from_inspected(&file, size, digest) != row.info - { - return Err(LtxError::ChecksumMismatch); - } - self.admit_segment_representation(&row.info, pages.len() * crate::paged::ENTRY_BYTES)?; - let index_bytes = Bytes::from(crate::paged::encode_index_from_pages(&pages)?); - // Retain at most one canonical small-pack budget of verified native - // inputs. If a later row exceeds it, release every earlier frozen - // body and keep the shared bundle representation for the whole tail. - if independent { - match independent_bytes - .checked_add(packed::HEADER_BYTES) - .and_then(|size| size.checked_add(row.info.size_bytes)) - .and_then(|size| size.checked_add(index_bytes.len() as u64)) - .filter(|size| *size <= upload::SINGLE_PUT_BYTES) - { - Some(size) => independent_bytes = size, - None => { - independent = false; - for input in &mut inputs { - input.body = AppendBody::Bundle; - } - } - } - } - inputs.push(AppendInput { - info: row.info.clone(), - location: BodyLocation::Bundle { - digest: bundle_digest, - offset: row.offset, - }, - index: index_bytes, - body: if independent { - AppendBody::Frozen(bytes) - } else { - AppendBody::Bundle - }, - }); - } - let target = inputs - .last() - .map(|input| input.info.position()) - .ok_or(LtxError::TxNotAvailable)?; + prospective.extend( + bundle + .rows() + .iter() + .filter(|row| row.repository == repository && row.epoch == epoch) + .map(|row| { + SegmentDescriptor::bundled( + row.info.clone(), + [0; 32], + 0, + bundle.digest(), + row.offset, + ) + }), + ); self.validate_chain(&prospective, target)?; if usage == BundleUse::IndependentRecovery && !independent { // A long history can repeatedly update the same small page image. diff --git a/crates/cellule-ltx/src/replica/shared/mod.rs b/crates/cellule-ltx/src/replica/shared/mod.rs index 93b95aae..952d2f42 100644 --- a/crates/cellule-ltx/src/replica/shared/mod.rs +++ b/crates/cellule-ltx/src/replica/shared/mod.rs @@ -107,6 +107,34 @@ impl SharedCaptures { } } +impl super::RecoveryOverlay { + /// Conservative encoded-byte and row bounds for this Cell's shared input. + /// + /// Includes the one object header, every original scoped row/body and its + /// maximum index. This allocates no bodies and grants no verified root or + /// authority; use it to reserve working memory before preparation. + pub fn shared_input_upper_bound(&self) -> Result<(u64, usize)> { + let (repository, epoch) = + crate::bundle::cell_identity(&self.predecessor.cell, &self.predecessor.incarnation); + let mut count = 0_usize; + let mut upper = HEADER_BYTES; + for row in self + .bundle + .rows() + .iter() + .filter(|row| row.repository == repository && row.epoch == epoch) + { + count += 1; + upper = upper + .checked_add(SCOPE_BYTES + packed::HEADER_BYTES) + .and_then(|bytes| bytes.checked_add(row.info.size_bytes)) + .and_then(|bytes| bytes.checked_add(u64::from(row.info.database_pages) * 60)) + .ok_or(LtxError::Limit(crate::LimitKind::CellBundleBytes))?; + } + Ok((upper, count)) + } +} + /// One Cell's verified inputs from a bounded publication cohort. /// /// Multi-row inputs contain uploaded shared extents. A singleton retains its @@ -141,6 +169,42 @@ impl CellReplica { return Ok(None); } let inputs = self.prepare_captured_inputs(&cuts.segments).await?; + self.shared_inputs(inputs, cuts.position).await.map(Some) + } + + /// Verifies small recovered rows through the canonical bundle-input reader. + /// + /// Original scopes and endpoint must match the overlay. Every original cut + /// remains in the returned input's admission facts before coalescing; the + /// later canonical root factory verifies the fresh base and complete chain. + /// Large tails return `None` for ordinary recovery preparation. The caller + /// retains memory and artifact admission through this operation and upload. + pub async fn shared_recovered_captures( + &self, + overlay: &super::RecoveryOverlay, + ) -> Result> { + self.validate_recovery_overlay(overlay)?; + if overlay.bundle.len() > self.limits.max_plan_bytes { + return Err(LtxError::Limit(crate::LimitKind::CellBundleBytes)); + } + let (upper, count) = overlay.shared_input_upper_bound()?; + if count > SHARED_PUBLICATION_ROWS || upper > SINGLE_PUT_BYTES { + return Ok(None); + } + let inputs = self.read_bundle_inputs(&overlay.bundle, true)?; + if !inputs.independent || inputs.target != overlay.final_position { + return Err(LtxError::ChecksumMismatch); + } + self.shared_inputs(inputs.inputs, inputs.target) + .await + .map(Some) + } + + async fn shared_inputs( + &self, + inputs: Vec, + position: Position, + ) -> Result { let segments: Vec<_> = inputs .into_iter() .map(|input| PreparedSegment { @@ -165,13 +229,13 @@ impl CellReplica { .and_then(|bytes| bytes.checked_add(segment.descriptor.index_length)) .ok_or(LtxError::LTXCorrupted) })?; - Ok(Some(SharedCaptures { + Ok(SharedCaptures { replica: self.clone(), - position: cuts.position, + position, segments, original, encoded_bytes, - })) + }) } /// Uploads a multi-row cohort once and returns independently scoped inputs. diff --git a/crates/cellule-ltx/tests/cell/roots.rs b/crates/cellule-ltx/tests/cell/roots.rs index 97d15505..908abe22 100644 --- a/crates/cellule-ltx/tests/cell/roots.rs +++ b/crates/cellule-ltx/tests/cell/roots.rs @@ -42,4 +42,5 @@ mod packed; mod preparation; mod prepare_cost; mod shared; +mod shared_recovery; mod sparse; diff --git a/crates/cellule-ltx/tests/cell/roots/lifecycle.rs b/crates/cellule-ltx/tests/cell/roots/lifecycle.rs index bad94eb6..1334c0a4 100644 --- a/crates/cellule-ltx/tests/cell/roots/lifecycle.rs +++ b/crates/cellule-ltx/tests/cell/roots/lifecycle.rs @@ -849,6 +849,13 @@ async fn recovered_overlay_keeps_the_bundle_when_a_later_row_exceeds_the_pack_bu ) .unwrap(); let overlay = RecoveryOverlay::new(base, bundle, last.position, 3); + assert!( + replica + .shared_recovered_captures(&overlay) + .await + .unwrap() + .is_none() + ); let recovered = replica .prepare_recovered_overlay(&overlay, 1) .await diff --git a/crates/cellule-ltx/tests/cell/roots/shared_recovery.rs b/crates/cellule-ltx/tests/cell/roots/shared_recovery.rs new file mode 100644 index 00000000..f495725a --- /dev/null +++ b/crates/cellule-ltx/tests/cell/roots/shared_recovery.rs @@ -0,0 +1,176 @@ +use super::*; + +fn bundle(cells: &[([u8; 32], [u8; 16], CaptureBatch)]) -> Bundle { + Bundle::encode( + cells + .iter() + .flat_map(|(cell, incarnation, cuts)| { + cuts.segments.iter().map(|segment| { + BundleEntry::for_cell( + *cell, + *incarnation, + segment.info().clone(), + std::fs::read(segment.path()).unwrap(), + ) + }) + }) + .collect(), + Limits::default(), + ) + .unwrap() +} + +#[tokio::test] +async fn shared_recovery_preserves_exact_scopes_original_chains_and_cold_bytes() { + let directory = tempfile::tempdir().unwrap(); + let counted = Arc::new(cellule_store::test_support::CountingObjectStore::new( + Arc::new(InMemory::new()), + )); + let store = Store::new(counted.clone()); + let mut replicas = Vec::new(); + let mut bases = Vec::new(); + let mut cells = Vec::new(); + for i in 0..2_u8 { + let cell = [91 + i; 32]; + let incarnation = [93 + i; 16]; + let replica = replica(store.clone(), cell, incarnation); + let mut db = Db::open( + &directory.path().join(format!("source-{i}")), + Limits::default(), + ) + .unwrap(); + db.transaction(|tx| tx.execute_batch("CREATE TABLE t(v); INSERT INTO t VALUES(1)")) + .unwrap(); + bases.push( + replica + .prepare(None, &db.capture().unwrap(), 1, 1) + .await + .unwrap() + .root(), + ); + db.transaction(|tx| tx.execute("INSERT INTO t VALUES(2)", [])) + .unwrap(); + let mut cuts = db.capture().unwrap(); + db.transaction(|tx| tx.execute("INSERT INTO t VALUES(3)", [])) + .unwrap(); + let last = db.capture().unwrap(); + cuts.segments.extend(last.segments); + cuts.position = last.position; + cells.push((cell, incarnation, cuts)); + replicas.push(replica); + db.close().unwrap(); + } + let mut captures = Vec::new(); + for (i, replica) in replicas.iter().enumerate() { + let overlay = RecoveryOverlay::new(bases[i], bundle(&cells), cells[i].2.position, 3); + let (_, rows) = overlay.shared_input_upper_bound().unwrap(); + assert_eq!(rows, 2, "only this Cell's original rows are charged"); + captures.push( + replica + .shared_recovered_captures(&overlay) + .await + .unwrap() + .unwrap(), + ); + } + counted.reset(); + let appends = CellReplica::upload_shared(captures, directory.path()) + .await + .unwrap(); + assert_eq!(counted.put_requests(), 1, "one shared immutable body"); + assert!( + replicas[1] + .prepare_shared(Some(&bases[1]), &appends[0], 3, 1) + .await + .is_err() + ); + for (i, replica) in replicas.iter().enumerate() { + let shared = replica + .prepare_shared(Some(&bases[i]), &appends[i], 3, 1) + .await + .unwrap(); + let overlay = RecoveryOverlay::new(bases[i], bundle(&cells), cells[i].2.position, 3); + let canonical = replica + .prepare_recovered_overlay(&overlay, 1) + .await + .unwrap(); + assert_eq!(shared.predecessor(), Some(bases[i])); + assert_eq!(shared.root().position, canonical.root().position); + assert_eq!(shared.root().commit_sequence, 3); + let cold = super::replica(store.clone(), cells[i].0, cells[i].1); + let shared_path = directory.path().join(format!("shared-{i}")); + let direct_path = directory.path().join(format!("direct-{i}")); + cold.open_root(&shared.root()) + .await + .unwrap() + .restore(&shared_path) + .await + .unwrap(); + cold.open_root(&canonical.root()) + .await + .unwrap() + .restore(&direct_path) + .await + .unwrap(); + assert_eq!( + std::fs::read(shared_path).unwrap(), + std::fs::read(direct_path).unwrap() + ); + + let limited = CellReplica::new( + CellStorageLayout::new(store.clone(), Path::from("runtime"), [3; 16]), + cells[i].0, + cells[i].1, + Limits { + max_segments: 2, + ..Limits::default() + }, + ) + .unwrap(); + counted.reset(); + assert!( + limited + .prepare_shared(Some(&bases[i]), &appends[i], 3, 1) + .await + .is_err(), + "one base plus two original cuts exceeds admission despite coalescing" + ); + assert_eq!(counted.put_requests(), 0); + } +} + +#[tokio::test] +async fn shared_recovery_rejects_overlay_scope_and_endpoint_before_upload() { + let directory = tempfile::tempdir().unwrap(); + let counted = Arc::new(cellule_store::test_support::CountingObjectStore::new( + Arc::new(InMemory::new()), + )); + let replica = replica(Store::new(counted.clone()), [97; 32], [98; 16]); + let mut db = Db::open(&directory.path().join("source"), Limits::default()).unwrap(); + db.transaction(|tx| tx.execute_batch("CREATE TABLE t(v); INSERT INTO t VALUES(1)")) + .unwrap(); + let base = replica + .prepare(None, &db.capture().unwrap(), 1, 1) + .await + .unwrap() + .root(); + db.transaction(|tx| tx.execute("INSERT INTO t VALUES(2)", [])) + .unwrap(); + let cells = vec![([97; 32], [98; 16], db.capture().unwrap())]; + for (predecessor, endpoint) in [ + ( + RootRef { + cell: [99; 32], + ..base + }, + cells[0].2.position, + ), + (base, base.position), + ] { + counted.reset(); + let overlay = RecoveryOverlay::new(predecessor, bundle(&cells), endpoint, 2); + assert!(replica.shared_recovered_captures(&overlay).await.is_err()); + assert_eq!(counted.put_requests(), 0); + } + db.close().unwrap(); +} diff --git a/crates/cellule-runtime/docs/write-performance-design.md b/crates/cellule-runtime/docs/write-performance-design.md index 9127df86..03392a2b 100644 --- a/crates/cellule-runtime/docs/write-performance-design.md +++ b/crates/cellule-runtime/docs/write-performance-design.md @@ -133,6 +133,15 @@ selection cohorts serially. Publication staging, reduced repeated verification and complete application performance qualification remain required. Component passes establish no new TPS claim. +Selected bundle materialization now submits eligible recovered rows to the same +shared root producer used by native captures. The canonical reader verifies +original scoped rows before coalescing; the root factory still checks the fresh +predecessor, complete original chain, lineage and exact endpoint before Cell CAS. +Pre-admitted memory follows dispatched native jobs through cancellation. The +original 256-KiB/64-row bounds remain; large tails or retained-memory pressure +use direct recovery preparation. End-to-end TPS and total GET/PUT bytes must +establish whether this integration improves application performance. + ```mermaid flowchart LR A[Bounded admission before SQL] --> B[Mutation and retry result commit together] diff --git a/crates/cellule-runtime/src/node/bundle/tests/materialization_shared.rs b/crates/cellule-runtime/src/node/bundle/tests/materialization_shared.rs new file mode 100644 index 00000000..8ee77d60 --- /dev/null +++ b/crates/cellule-runtime/src/node/bundle/tests/materialization_shared.rs @@ -0,0 +1,180 @@ +use super::*; +use crate::fleet::resource::{ResourceCost, ResourceLedger}; +use crate::fleet::telemetry::{CellTelemetry, CellTelemetryHandle, SharedPublicationTiming}; +use crate::publication::SharedPublication; +use futures_util::{StreamExt as _, stream::FuturesUnordered}; +use std::sync::atomic::{AtomicU64, Ordering}; + +#[derive(Default)] +struct SharedEvidence { + cells: AtomicU64, + pressure: AtomicU64, +} + +impl CellTelemetry for SharedEvidence { + fn shared_publication(&self, timing: SharedPublicationTiming) { + assert!(timing.succeeded); + self.cells.fetch_add(timing.cells, Ordering::SeqCst); + } + + fn shared_publication_singleton(&self, _: std::time::Duration) { + self.cells.fetch_add(1, Ordering::SeqCst); + } + + fn shared_publication_fallback(&self, pressure: bool) { + assert!( + pressure, + "these small verified overlays fit the original byte bound" + ); + self.pressure.fetch_add(1, Ordering::SeqCst); + } +} + +#[tokio::test] +async fn selected_bundle_materialization_enters_shared_producer_and_recovers_exactly() { + materialize(8 << 20, 8, 0).await; +} + +#[tokio::test] +async fn selected_bundle_materialization_falls_back_under_original_memory_pressure() { + materialize(1, 0, 8).await; +} + +async fn materialize(memory: usize, shared_cells: u64, pressure: u64) { + let mut f = Fixture::new().await; + let host = cellule_ltx::Host::default() + .with_local_disk_budget(cellule_ltx::DiskBudget::new(8 << 20)) + .with_io_slots(Arc::new(tokio::sync::Semaphore::new(1))) + .with_job_slots(Arc::new(tokio::sync::Semaphore::new(1))) + .with_dirty_slots(Arc::new(tokio::sync::Semaphore::new(1))) + .with_recovery_slots(Arc::new(tokio::sync::Semaphore::new(1))) + .with_scratch_slots(Arc::new(tokio::sync::Semaphore::new(1))); + let disk = host.local_disk_budget(); + let ledger = ResourceLedger::new( + ResourceCost::zero() + .with_retained_bytes(memory) + .with_publication_file_descriptors(64), + ); + let evidence = Arc::new(SharedEvidence::default()); + let coordinator = SharedPublication::new( + ledger.clone(), + CellTelemetryHandle::from_sink(evidence.clone()), + ); + let mut cells = Vec::new(); + let mut frames = Vec::new(); + let mut assignments = Vec::new(); + for byte in 4..12 { + let mut cell = f.cell(byte).await; + cell.replica = cell.replica.with_host(host.clone()); + let (_, next, assigned) = f.append(&mut cell, 2); + frames.extend(next); + assignments.push(assigned); + cells.push(cell); + } + let prepared = f + .directory + .prepare_node_bundle(&f.node, &frames, &assignments, NOW) + .await + .unwrap(); + let (node, proofs) = f + .directory + .select_node_bundle(&f.node, &prepared, &f.lease, Limits::default(), NOW) + .await + .unwrap(); + f.node = node; + let mut tasks = FuturesUnordered::new(); + for cell in &cells { + let proof = proofs + .iter() + .find(|proof| proof.binding.control.cell == cell.control.value().cell) + .unwrap(); + let mut publisher = f + .publisher(cell) + .with_shared_publication(coordinator.clone()); + let expected_cell = cell.control.value().cell; + tasks.push(async move { + let root = publisher.materialize_bundle(proof).await.unwrap(); + (expected_cell, root) + }); + } + let mut roots = Vec::new(); + while let Some(result) = tokio::time::timeout(std::time::Duration::from_secs(5), tasks.next()) + .await + .unwrap() + { + roots.push(result); + } + assert_eq!(roots.len(), 8); + coordinator.shutdown().await.unwrap(); + assert_eq!(ledger.snapshot().unwrap().used, ResourceCost::zero()); + assert_eq!(disk.used(), 0); + for (cell, root) in roots { + assert_eq!(root.commit_sequence, 2); + let original = cells + .iter() + .find(|candidate| candidate.control.value().cell == cell) + .unwrap(); + let proof = proofs + .iter() + .find(|proof| proof.binding.control.cell == cell) + .unwrap(); + let overlay = proof + .recovery_overlay(original.authority.layout(), Limits::default()) + .await + .unwrap(); + let direct = original + .replica + .prepare_recovered_overlay(&overlay, 1) + .await + .unwrap(); + let cold = CellReplica::new( + original.authority.layout().clone(), + root.cell, + root.incarnation, + Limits::default(), + ) + .unwrap(); + cold.reachable_objects(&root).await.unwrap(); + let destination = f + .scratch + .path() + .join(format!("shared-{}", cell.as_bytes()[0])); + let canonical = f + .scratch + .path() + .join(format!("direct-{}", cell.as_bytes()[0])); + cold.open_root(&root) + .await + .unwrap() + .restore(&destination) + .await + .unwrap(); + cold.open_root(&direct.root()) + .await + .unwrap() + .restore(&canonical) + .await + .unwrap(); + assert_eq!( + std::fs::read(&destination).unwrap(), + std::fs::read(canonical).unwrap() + ); + let db = rusqlite::Connection::open(destination).unwrap(); + let outcomes: Vec<(String, String)> = db + .prepare("SELECT request,result FROM outcomes ORDER BY request") + .unwrap() + .query_map([], |row| Ok((row.get(0)?, row.get(1)?))) + .unwrap() + .collect::>() + .unwrap(); + assert_eq!( + outcomes, + [ + ("request-2".into(), "result-2".into()), + ("seed".into(), "original".into()) + ] + ); + } + assert_eq!(evidence.cells.load(Ordering::SeqCst), shared_cells); + assert_eq!(evidence.pressure.load(Ordering::SeqCst), pressure); +} diff --git a/crates/cellule-runtime/src/node/bundle/tests/mod.rs b/crates/cellule-runtime/src/node/bundle/tests/mod.rs index ecd872ce..68ed64db 100644 --- a/crates/cellule-runtime/src/node/bundle/tests/mod.rs +++ b/crates/cellule-runtime/src/node/bundle/tests/mod.rs @@ -21,6 +21,7 @@ mod faults; mod index; mod lifecycle; mod managed; +mod materialization_shared; mod ranges; mod readiness; mod receipt_pressure; diff --git a/crates/cellule-runtime/src/publication/mod.rs b/crates/cellule-runtime/src/publication/mod.rs index ff412d0b..b990d5fb 100644 --- a/crates/cellule-runtime/src/publication/mod.rs +++ b/crates/cellule-runtime/src/publication/mod.rs @@ -11,6 +11,7 @@ use crate::retry::{Backoff, retry_hint, retryable_storage_error}; use crate::{Error, Result}; pub(crate) mod lineage; +mod recovered; mod shared; pub(crate) use shared::{PublicationPermit, SharedPublication}; @@ -129,14 +130,7 @@ impl CellPublisher { .recovery_overlay_from(self.authority.layout(), self.replica.limits(), base) .await?; self.check_node_lease()?; - let (replica, confirmation) = lineage::replica(self.replica.clone(), &self.authority); - let prepared = replica - .prepare_recovered_overlay(&overlay, self.observed.value().schema) - .await - .map_err(lineage::error)?; - self.lineage_confirmed = *confirmation - .lock() - .map_err(|_| Error::Peer("root lineage confirmation lock poisoned"))?; + let prepared = self.prepare_recovered(&overlay).await?; self.publish_prepared(&prepared, next_due_ms).await } diff --git a/crates/cellule-runtime/src/publication/recovered.rs b/crates/cellule-runtime/src/publication/recovered.rs new file mode 100644 index 00000000..a135c292 --- /dev/null +++ b/crates/cellule-runtime/src/publication/recovered.rs @@ -0,0 +1,88 @@ +//! Selected overlays enter the same bounded producer and exact root factory. +use super::*; + +impl CellPublisher { + pub(super) async fn prepare_recovered( + &mut self, + overlay: &cellule_ltx::RecoveryOverlay, + ) -> Result { + let result = match self.admit_publication().await? { + PublicationPermit::Direct(replica) => { + self.prepare_recovered_inputs(*replica, overlay, None).await + } + PublicationPermit::Shared(slot) => { + let coordinator = self + .shared_publication + .clone() + .ok_or(Error::Control("shared publication is unavailable"))?; + let replica = self.replica.clone(); + let submission = coordinator.submit_recovered( + &replica, + overlay, + self.scratch_directory.clone(), + slot, + ); + tokio::pin!(submission); + let shared = loop { + tokio::select! { + result = &mut submission => break result, + _ = tokio::time::sleep_until(tokio::time::Instant::from_std(self.renew_at)) => self.renew().await?, + } + }; + self.check_node_lease()?; + let shared = match shared { + Ok(shared) => shared, + // A failed sibling/upload cannot invalidate this Cell's + // original overlay. The direct factory rechecks its exact + // inputs and retains its own source error on failure. + Err(Error::Shared(source)) if matches!(source.as_ref(), Error::Ltx(_)) => None, + Err(error) => return Err(error), + }; + self.prepare_recovered_inputs(self.replica.clone(), overlay, shared.as_ref()) + .await + } + }; + self.record_publication_cost(); + result + } + + async fn prepare_recovered_inputs( + &mut self, + replica: cellule_ltx::CellReplica, + overlay: &cellule_ltx::RecoveryOverlay, + shared: Option<&shared::SharedPrepared>, + ) -> Result { + self.check_node_lease()?; + let (replica, confirmation) = lineage::replica(replica, &self.authority); + let base = overlay.predecessor(); + let schema = self.observed.value().schema; + let preparation = match shared { + Some(shared) => futures_util::future::Either::Left(replica.prepare_shared( + Some(&base), + &shared.append, + overlay.final_commit_sequence(), + schema, + )), + None => futures_util::future::Either::Right( + replica.prepare_recovered_overlay(overlay, schema), + ), + }; + tokio::pin!(preparation); + let prepared = loop { + tokio::select! { + result = &mut preparation => break result.map_err(lineage::error)?, + _ = tokio::time::sleep_until(tokio::time::Instant::from_std(self.renew_at)) => self.renew().await?, + } + }; + if prepared.predecessor() != Some(base) + || prepared.root().position != overlay.final_position() + || prepared.root().commit_sequence != overlay.final_commit_sequence() + { + return Err(cellule_ltx::LtxError::ChecksumMismatch.into()); + } + self.lineage_confirmed = *confirmation + .lock() + .map_err(|_| Error::Peer("root lineage confirmation lock poisoned"))?; + Ok(prepared) + } +} diff --git a/crates/cellule-runtime/src/publication/shared/mod.rs b/crates/cellule-runtime/src/publication/shared/mod.rs index 84cc82e0..7547943d 100644 --- a/crates/cellule-runtime/src/publication/shared/mod.rs +++ b/crates/cellule-runtime/src/publication/shared/mod.rs @@ -30,14 +30,20 @@ pub(crate) enum PublicationPermit { pub(crate) struct SharedPrepared { pub(crate) append: SharedAppend, // Index/row tables remain charged through per-Cell root preparation. - _memory: ResourceReservation, + _memory: Arc, } +struct SharedMemory { + _reservation: ResourceReservation, +} + +impl cellule_ltx::HostResourcePermit for SharedMemory {} + struct Entry { captures: SharedCaptures, scratch: PathBuf, accepted_at: Instant, - memory: ResourceReservation, + memory: Arc, _slot: OwnedSemaphorePermit, reply: oneshot::Sender>, } @@ -100,15 +106,60 @@ impl SharedPublication { .checked_add(row) .ok_or(Error::Capacity("shared publication bytes")) })?; - if estimate as u64 > SHARED_PUBLICATION_BYTES - || cuts.segments.len() > SHARED_PUBLICATION_ROWS - { + let Some(memory) = + self.reserve_input(estimate, cuts.segments.len(), cuts.segments.len() + 2)? + else { + return Ok(None); + }; + let replica = replica.clone().with_preparation_resource(memory.clone()); + let Some(captures) = replica.shared_captures(cuts).await? else { + self.telemetry.shared_publication_fallback(false); + return Ok(None); + }; + self.submit_captures(captures, memory, scratch, slot) + .await + .map(Some) + } + + pub(crate) async fn submit_recovered( + &self, + replica: &cellule_ltx::CellReplica, + overlay: &cellule_ltx::RecoveryOverlay, + scratch: PathBuf, + slot: OwnedSemaphorePermit, + ) -> Result> { + let (estimate, rows) = overlay.shared_input_upper_bound()?; + let estimate = + usize::try_from(estimate).map_err(|_| Error::Capacity("shared publication bytes"))?; + // The enclosing overlay owns its original artifact. Verified frozen + // rows need only the bounded shared scratch descriptors, not one native + // file pin per row. Never retain a Cell dirty slot while enqueued. + let Some(memory) = self.reserve_input(estimate, rows, 2)? else { + return Ok(None); + }; + let replica = replica.clone().with_preparation_resource(memory.clone()); + let Some(captures) = replica.shared_recovered_captures(overlay).await? else { + self.telemetry.shared_publication_fallback(false); + return Ok(None); + }; + self.submit_captures(captures, memory, scratch, slot) + .await + .map(Some) + } + + fn reserve_input( + &self, + estimate: usize, + rows: usize, + file_descriptors: usize, + ) -> Result>> { + if estimate as u64 > SHARED_PUBLICATION_BYTES || rows > SHARED_PUBLICATION_ROWS { self.telemetry.shared_publication_fallback(false); return Ok(None); } // Charge before pinning/indexing/coalescing. Multiple compressed cuts // can expand into a bounded 256 KiB page map before being re-encoded. - let work = if cuts.segments.len() > 1 { + let work = if rows > 1 { estimate.max(SHARED_PUBLICATION_BYTES as usize) } else { estimate @@ -119,7 +170,7 @@ impl SharedPublication { .ok_or(Error::Capacity("shared publication memory"))?; let cost = ResourceCost::zero() .with_retained_bytes(memory) - .with_publication_file_descriptors(cuts.segments.len() + 2); + .with_publication_file_descriptors(file_descriptors); let memory = match self.resources.try_reserve(cost) { Ok(memory) => memory, // Sharing is optional representation reduction. Never wait for @@ -130,10 +181,18 @@ impl SharedPublication { } Err(error) => return Err(error), }; - let Some(captures) = replica.shared_captures(cuts).await? else { - self.telemetry.shared_publication_fallback(false); - return Ok(None); - }; + Ok(Some(Arc::new(SharedMemory { + _reservation: memory, + }))) + } + + async fn submit_captures( + &self, + captures: SharedCaptures, + memory: Arc, + scratch: PathBuf, + slot: OwnedSemaphorePermit, + ) -> Result { let (reply, response) = oneshot::channel(); self.sender .send(Message::Capture(Box::new(Entry { @@ -147,7 +206,7 @@ impl SharedPublication { .await .map_err(|_| Error::RuntimeClosed)?; // The lane, rather than the waiter, owns dispatched storage and scratch. - response.await.map_err(|_| Error::RuntimeClosed)?.map(Some) + response.await.map_err(|_| Error::RuntimeClosed)? } pub(crate) async fn shutdown(&self) -> Result<()> { diff --git a/crates/cellule-runtime/src/publication/shared/tests.rs b/crates/cellule-runtime/src/publication/shared/tests.rs index 0f127244..69c09d80 100644 --- a/crates/cellule-runtime/src/publication/shared/tests.rs +++ b/crates/cellule-runtime/src/publication/shared/tests.rs @@ -114,7 +114,9 @@ async fn minimum_host_permits_drain_shared_work_after_a_sibling_waiter_cancels() captures, scratch: directory.path().to_owned(), accepted_at: Instant::now(), - memory, + memory: Arc::new(SharedMemory { + _reservation: memory, + }), _slot: slot, reply, }); From ce3e8d6dbc7b6607c8b9675d61118409d9edeb90 Mon Sep 17 00:00:00 2001 From: forhappy Date: Fri, 9 Oct 2026 17:10:57 -0700 Subject: [PATCH 097/102] Revert "Route recovered overlays through bounded shared publication" This reverts commit c2c0ef10b08b9844401ca9e11397e9321a744f91. --- crates/cellule-ltx/docs/publication.md | 13 -- .../cellule-ltx/src/environment/host/mod.rs | 18 -- .../cellule-ltx/src/environment/host/tests.rs | 69 ------- .../cellule-ltx/src/replica/bundle_inputs.rs | 109 ----------- crates/cellule-ltx/src/replica/mod.rs | 15 -- crates/cellule-ltx/src/replica/prepare.rs | 109 +++++++++-- crates/cellule-ltx/src/replica/shared/mod.rs | 70 +------ crates/cellule-ltx/tests/cell/roots.rs | 1 - .../cellule-ltx/tests/cell/roots/lifecycle.rs | 7 - .../tests/cell/roots/shared_recovery.rs | 176 ----------------- .../docs/write-performance-design.md | 9 - .../bundle/tests/materialization_shared.rs | 180 ------------------ .../src/node/bundle/tests/mod.rs | 1 - crates/cellule-runtime/src/publication/mod.rs | 10 +- .../src/publication/recovered.rs | 88 --------- .../src/publication/shared/mod.rs | 83 ++------ .../src/publication/shared/tests.rs | 4 +- 17 files changed, 112 insertions(+), 850 deletions(-) delete mode 100644 crates/cellule-ltx/src/environment/host/tests.rs delete mode 100644 crates/cellule-ltx/src/replica/bundle_inputs.rs delete mode 100644 crates/cellule-ltx/tests/cell/roots/shared_recovery.rs delete mode 100644 crates/cellule-runtime/src/node/bundle/tests/materialization_shared.rs delete mode 100644 crates/cellule-runtime/src/publication/recovered.rs diff --git a/crates/cellule-ltx/docs/publication.md b/crates/cellule-ltx/docs/publication.md index 3b7898f9..32668a53 100644 --- a/crates/cellule-ltx/docs/publication.md +++ b/crates/cellule-ltx/docs/publication.md @@ -23,9 +23,6 @@ sequenceDiagram | `CellReplica::prepare` | Verifies cuts and writes immutable root dependencies. | | `prepare_bundle` | Selects this Cell's exact rows from a shared bundle. | | `prepare_recovered_overlay` | Verifies the exact predecessor and final position; small independent tails use the canonical coalescer and native pack. | -| `shared_recovered_captures` | Verifies eligible overlay rows through the same input reader; original chain facts survive coalescing for the canonical shared root factory. | -| `RecoveryOverlay::shared_input_upper_bound` | Provides conservative original row/body/index bounds before allocation; grants no verification or authority. | -| `with_preparation_resource` | Keeps a caller's existing memory/artifact admission alive in dispatched jobs after waiter cancellation; reserves nothing and grants no authority. | | `prepare_compaction` | Rewrites representation without changing logical state. | | `prepare_after_compaction` | Appends to a private compaction while retaining its original authority predecessor. | | `try_admit_scheduled_compaction` | Returns a scoped clone with existing dirty/recovery admission, or defers without waiting behind a queued cohort. | @@ -91,16 +88,6 @@ individual row, changed image or output retains the ordinary bundle path. `prepare_bundle` preserves shared bundle references. No root format, authority rule or host resource ceiling changes. -Selected bundle materialization can feed those same verified small inputs into -the runtime's existing shared publication producer. The original 256-KiB object -and 64-row bounds apply before reduction; larger tails retain ordinary recovery -preparation. Retained-memory pressure selects that direct fallback immediately, -without holding a dirty permit while waiting for the shared upload. Each Cell -still verifies its fresh predecessor, complete original chain and exact endpoint, -retains native lineage, then selects its own root under the existing fenced CAS. -Shared-object origin verification remains complete; lower upload counts alone -do not establish lower total publication bytes or application TPS. - A representation-only compaction can remain private while its successor append uploads. `prepare_after_compaction` verifies that the compaction preserves the predecessor's position, commit sequence, Cell and incarnation. The runtime selects diff --git a/crates/cellule-ltx/src/environment/host/mod.rs b/crates/cellule-ltx/src/environment/host/mod.rs index 7b7aad1a..1e22d298 100644 --- a/crates/cellule-ltx/src/environment/host/mod.rs +++ b/crates/cellule-ltx/src/environment/host/mod.rs @@ -14,9 +14,6 @@ use std::{ mod admission; mod budget; -#[cfg(all(test, feature = "replica"))] -mod tests; - pub use budget::{DiskBudget, DiskReservation}; #[cfg(feature = "replica")] @@ -203,8 +200,6 @@ pub struct Host { dirty: Option>, #[cfg(feature = "replica")] scratch: Option>, - #[cfg(feature = "replica")] - preparation_resource: Option>, } #[cfg(feature = "replica")] @@ -729,15 +724,6 @@ impl Host { .map_err(crate::LtxError::Io) } - #[cfg(feature = "replica")] - pub(crate) fn with_preparation_resource( - mut self, - resource: Arc, - ) -> Self { - self.preparation_resource = Some(resource); - self - } - #[cfg(feature = "replica")] pub(crate) async fn run( &self, @@ -763,7 +749,6 @@ impl Host { let recovery = self.recovery.clone(); let dirty = self.dirty.clone(); let scratch = self.scratch.clone(); - let preparation_resource = self.preparation_resource.clone(); self.executor.dispatch(Box::new(move || { // Dispatched work can outlive its future. Keep admission with the // job, not the waiter, so cancellation cannot oversubscribe the pool. @@ -775,7 +760,6 @@ impl Host { drop(recovery); drop(dirty); drop(scratch); - drop(preparation_resource); drop(resource); drop(permit); let _ = send.send(result); @@ -872,8 +856,6 @@ impl Default for Host { dirty: None, #[cfg(feature = "replica")] scratch: None, - #[cfg(feature = "replica")] - preparation_resource: None, } } } diff --git a/crates/cellule-ltx/src/environment/host/tests.rs b/crates/cellule-ltx/src/environment/host/tests.rs deleted file mode 100644 index 1e9b913f..00000000 --- a/crates/cellule-ltx/src/environment/host/tests.rs +++ /dev/null @@ -1,69 +0,0 @@ -use super::*; -use crate::environment::executor::Worker; - -#[derive(Default)] -struct HeldExecutor { - jobs: Mutex>>, - started: tokio::sync::Notify, -} - -impl Executor for HeldExecutor { - fn dispatch(&self, job: Box) -> io::Result<()> { - self.jobs.lock().unwrap().push(job); - self.started.notify_one(); - Ok(()) - } - - fn start_worker(&self, job: Box) -> io::Result> { - TokioExecutor.start_worker(job) - } -} - -struct PreparationResource(Arc); - -impl HostResourcePermit for PreparationResource {} - -impl Drop for PreparationResource { - fn drop(&mut self) { - self.0.store(true, Ordering::SeqCst); - } -} - -#[tokio::test] -async fn preparation_resources_survive_cancelled_native_waiters_and_release_on_completion() { - for mode in 0..3 { - let executor = Arc::new(HeldExecutor::default()); - let jobs = Arc::new(tokio::sync::Semaphore::new(1)); - let released = Arc::new(AtomicBool::new(false)); - let host = Host::default() - .with_executor(executor.clone()) - .with_job_slots(jobs.clone()) - .with_preparation_resource(Arc::new(PreparationResource(released.clone()))); - let mut work = Box::pin(async move { - host.run(move || { - assert_ne!(mode, 2, "injected native failure"); - 7 - }) - .await - }); - tokio::time::timeout(Duration::from_secs(1), async { - tokio::select! { - result = &mut work => panic!("held original job completed: {result:?}"), - _ = executor.started.notified() => {} - } - }) - .await - .unwrap(); - drop(work); - assert!(!released.load(Ordering::SeqCst)); - assert_eq!(jobs.available_permits(), 0); - let job = executor.jobs.lock().unwrap().pop().unwrap(); - if mode == 1 { - drop(job); - } else { - job(); - } - assert!(released.load(Ordering::SeqCst)); - assert_eq!(jobs.available_permits(), 1); - } -} diff --git a/crates/cellule-ltx/src/replica/bundle_inputs.rs b/crates/cellule-ltx/src/replica/bundle_inputs.rs deleted file mode 100644 index aa907f5b..00000000 --- a/crates/cellule-ltx/src/replica/bundle_inputs.rs +++ /dev/null @@ -1,109 +0,0 @@ -//! Canonical verified Cell inputs before bundle representation reduction. -use super::*; - -pub(super) struct BundleInputs { - pub inputs: Vec, - pub target: Position, - pub independent: bool, -} - -impl CellReplica { - pub(super) fn validate_recovery_overlay(&self, overlay: &RecoveryOverlay) -> Result<()> { - if overlay.predecessor.cell != self.cell - || overlay.predecessor.incarnation != self.incarnation - || overlay.final_commit_sequence <= overlay.predecessor.commit_sequence - { - return Err(LtxError::InvalidState("recovery overlay scope")); - } - let (repository, epoch) = crate::bundle::cell_identity(&self.cell, &self.incarnation); - let final_position = overlay - .bundle - .rows() - .iter() - .rfind(|row| row.repository == repository && row.epoch == epoch) - .map(|row| row.info.position()) - .ok_or(LtxError::TxNotAvailable)?; - if final_position != overlay.final_position { - return Err(LtxError::ChecksumMismatch); - } - Ok(()) - } - - pub(super) fn read_bundle_inputs( - &self, - bundle: &crate::bundle::Bundle, - mut independent: bool, - ) -> Result { - if bundle.len() > self.limits.max_plan_bytes { - return Err(LtxError::Limit(crate::LimitKind::CellBundleBytes)); - } - let (repository, epoch) = crate::bundle::cell_identity(&self.cell, &self.incarnation); - let bundle_digest = bundle.digest(); - let mut inputs: Vec = Vec::new(); - let mut selected_bytes = 0_u64; - let mut independent_bytes = 0_u64; - for (index, row) in bundle.rows().iter().enumerate() { - if row.repository != repository || row.epoch != epoch { - continue; - } - selected_bytes = selected_bytes - .checked_add(row.info.size_bytes) - .ok_or(LtxError::Limit(crate::LimitKind::CapturedCellBundleBytes))?; - if row.info.size_bytes > self.limits.max_file_bytes - || selected_bytes > self.limits.max_plan_bytes - { - return Err(LtxError::Limit(crate::LimitKind::CapturedCellBundleBytes)); - } - let bytes = bundle.read_segment(index)?; - let (file, size, digest, pages) = crate::ltx::inspect_bytes_with_index(&bytes)?; - if size != row.info.size_bytes - || digest != row.info.blake3 - || crate::SegmentInfo::from_inspected(&file, size, digest) != row.info - { - return Err(LtxError::ChecksumMismatch); - } - self.admit_segment_representation(&row.info, pages.len() * crate::paged::ENTRY_BYTES)?; - let index_bytes = Bytes::from(crate::paged::encode_index_from_pages(&pages)?); - // Release earlier frozen rows as soon as the original small-pack - // allowance fails. A shared producer cannot widen this bound. - if independent { - match independent_bytes - .checked_add(packed::HEADER_BYTES) - .and_then(|size| size.checked_add(row.info.size_bytes)) - .and_then(|size| size.checked_add(index_bytes.len() as u64)) - .filter(|size| *size <= upload::SINGLE_PUT_BYTES) - { - Some(size) => independent_bytes = size, - None => { - independent = false; - for input in &mut inputs { - input.body = AppendBody::Bundle; - } - } - } - } - inputs.push(AppendInput { - info: row.info.clone(), - location: BodyLocation::Bundle { - digest: bundle_digest, - offset: row.offset, - }, - index: index_bytes, - body: if independent { - AppendBody::Frozen(bytes) - } else { - AppendBody::Bundle - }, - }); - } - let target = inputs - .last() - .map(|input| input.info.position()) - .ok_or(LtxError::TxNotAvailable)?; - Ok(BundleInputs { - inputs, - target, - independent, - }) - } -} diff --git a/crates/cellule-ltx/src/replica/mod.rs b/crates/cellule-ltx/src/replica/mod.rs index d9d5f984..384fa32e 100644 --- a/crates/cellule-ltx/src/replica/mod.rs +++ b/crates/cellule-ltx/src/replica/mod.rs @@ -13,7 +13,6 @@ use futures_util::{StreamExt as _, TryStreamExt as _, stream}; use crate::{CaptureBatch, Host, Limits, LtxError, Position, Result}; -mod bundle_inputs; mod cache; mod coalesce; mod compaction; @@ -823,20 +822,6 @@ impl CellReplica { self } - /// Retains a caller's pre-admitted preparation resources through native jobs. - /// - /// This reserves no resources and grants no root or writer authority. Use a - /// scoped clone: dispatched jobs keep the token after waiter cancellation, - /// and shared inputs keep it until their preparation owner releases them. - #[must_use] - pub fn with_preparation_resource( - mut self, - resource: Arc, - ) -> Self { - self.host = self.host.with_preparation_resource(resource); - self - } - /// Exclusively creates a fresh local database using this replica's host and limits. /// /// The destination and SQLite sidecars must not exist. A failed open leaves diff --git a/crates/cellule-ltx/src/replica/prepare.rs b/crates/cellule-ltx/src/replica/prepare.rs index 01ddf119..177e2115 100644 --- a/crates/cellule-ltx/src/replica/prepare.rs +++ b/crates/cellule-ltx/src/replica/prepare.rs @@ -236,7 +236,23 @@ impl CellReplica { overlay: &RecoveryOverlay, schema: u32, ) -> Result { - self.validate_recovery_overlay(overlay)?; + if overlay.predecessor.cell != self.cell + || overlay.predecessor.incarnation != self.incarnation + || overlay.final_commit_sequence <= overlay.predecessor.commit_sequence + { + return Err(LtxError::InvalidState("recovery overlay scope")); + } + let (repository, epoch) = crate::bundle::cell_identity(&self.cell, &self.incarnation); + let final_position = overlay + .bundle + .rows() + .iter() + .rfind(|row| row.repository == repository && row.epoch == epoch) + .map(|row| row.info.position()) + .ok_or(LtxError::TxNotAvailable)?; + if final_position != overlay.final_position { + return Err(LtxError::ChecksumMismatch); + } let mut replica = self.clone(); replica.host = self.host.for_dirty().await?; let prepared = replica @@ -273,30 +289,81 @@ impl CellReplica { self.validate_append_sequence(&base_graph, commit_sequence)?; let (repository, epoch) = crate::bundle::cell_identity(&self.cell, &self.incarnation); - let bundle_inputs = - self.read_bundle_inputs(bundle, usage == BundleUse::IndependentRecovery)?; - let mut inputs = bundle_inputs.inputs; - let target = bundle_inputs.target; - let mut independent = bundle_inputs.independent; + let bundle_digest = bundle.digest(); + let mut inputs: Vec = Vec::new(); + let mut selected_bytes = 0_u64; + let mut independent = usage == BundleUse::IndependentRecovery; + let mut independent_bytes = 0_u64; let mut prospective = base_graph .as_ref() .map(|graph| graph.descriptors.clone()) .unwrap_or_default(); - prospective.extend( - bundle - .rows() - .iter() - .filter(|row| row.repository == repository && row.epoch == epoch) - .map(|row| { - SegmentDescriptor::bundled( - row.info.clone(), - [0; 32], - 0, - bundle.digest(), - row.offset, - ) - }), - ); + for (index, row) in bundle.rows().iter().enumerate() { + if row.repository != repository || row.epoch != epoch { + continue; + } + selected_bytes = selected_bytes + .checked_add(row.info.size_bytes) + .ok_or(LtxError::Limit(crate::LimitKind::CapturedCellBundleBytes))?; + if row.info.size_bytes > self.limits.max_file_bytes + || selected_bytes > self.limits.max_plan_bytes + { + return Err(LtxError::Limit(crate::LimitKind::CapturedCellBundleBytes)); + } + prospective.push(SegmentDescriptor::bundled( + row.info.clone(), + [0; 32], + 0, + bundle_digest, + row.offset, + )); + let bytes = bundle.read_segment(index)?; + let (file, size, digest, pages) = crate::ltx::inspect_bytes_with_index(&bytes)?; + if size != row.info.size_bytes + || digest != row.info.blake3 + || crate::SegmentInfo::from_inspected(&file, size, digest) != row.info + { + return Err(LtxError::ChecksumMismatch); + } + self.admit_segment_representation(&row.info, pages.len() * crate::paged::ENTRY_BYTES)?; + let index_bytes = Bytes::from(crate::paged::encode_index_from_pages(&pages)?); + // Retain at most one canonical small-pack budget of verified native + // inputs. If a later row exceeds it, release every earlier frozen + // body and keep the shared bundle representation for the whole tail. + if independent { + match independent_bytes + .checked_add(packed::HEADER_BYTES) + .and_then(|size| size.checked_add(row.info.size_bytes)) + .and_then(|size| size.checked_add(index_bytes.len() as u64)) + .filter(|size| *size <= upload::SINGLE_PUT_BYTES) + { + Some(size) => independent_bytes = size, + None => { + independent = false; + for input in &mut inputs { + input.body = AppendBody::Bundle; + } + } + } + } + inputs.push(AppendInput { + info: row.info.clone(), + location: BodyLocation::Bundle { + digest: bundle_digest, + offset: row.offset, + }, + index: index_bytes, + body: if independent { + AppendBody::Frozen(bytes) + } else { + AppendBody::Bundle + }, + }); + } + let target = inputs + .last() + .map(|input| input.info.position()) + .ok_or(LtxError::TxNotAvailable)?; self.validate_chain(&prospective, target)?; if usage == BundleUse::IndependentRecovery && !independent { // A long history can repeatedly update the same small page image. diff --git a/crates/cellule-ltx/src/replica/shared/mod.rs b/crates/cellule-ltx/src/replica/shared/mod.rs index 952d2f42..93b95aae 100644 --- a/crates/cellule-ltx/src/replica/shared/mod.rs +++ b/crates/cellule-ltx/src/replica/shared/mod.rs @@ -107,34 +107,6 @@ impl SharedCaptures { } } -impl super::RecoveryOverlay { - /// Conservative encoded-byte and row bounds for this Cell's shared input. - /// - /// Includes the one object header, every original scoped row/body and its - /// maximum index. This allocates no bodies and grants no verified root or - /// authority; use it to reserve working memory before preparation. - pub fn shared_input_upper_bound(&self) -> Result<(u64, usize)> { - let (repository, epoch) = - crate::bundle::cell_identity(&self.predecessor.cell, &self.predecessor.incarnation); - let mut count = 0_usize; - let mut upper = HEADER_BYTES; - for row in self - .bundle - .rows() - .iter() - .filter(|row| row.repository == repository && row.epoch == epoch) - { - count += 1; - upper = upper - .checked_add(SCOPE_BYTES + packed::HEADER_BYTES) - .and_then(|bytes| bytes.checked_add(row.info.size_bytes)) - .and_then(|bytes| bytes.checked_add(u64::from(row.info.database_pages) * 60)) - .ok_or(LtxError::Limit(crate::LimitKind::CellBundleBytes))?; - } - Ok((upper, count)) - } -} - /// One Cell's verified inputs from a bounded publication cohort. /// /// Multi-row inputs contain uploaded shared extents. A singleton retains its @@ -169,42 +141,6 @@ impl CellReplica { return Ok(None); } let inputs = self.prepare_captured_inputs(&cuts.segments).await?; - self.shared_inputs(inputs, cuts.position).await.map(Some) - } - - /// Verifies small recovered rows through the canonical bundle-input reader. - /// - /// Original scopes and endpoint must match the overlay. Every original cut - /// remains in the returned input's admission facts before coalescing; the - /// later canonical root factory verifies the fresh base and complete chain. - /// Large tails return `None` for ordinary recovery preparation. The caller - /// retains memory and artifact admission through this operation and upload. - pub async fn shared_recovered_captures( - &self, - overlay: &super::RecoveryOverlay, - ) -> Result> { - self.validate_recovery_overlay(overlay)?; - if overlay.bundle.len() > self.limits.max_plan_bytes { - return Err(LtxError::Limit(crate::LimitKind::CellBundleBytes)); - } - let (upper, count) = overlay.shared_input_upper_bound()?; - if count > SHARED_PUBLICATION_ROWS || upper > SINGLE_PUT_BYTES { - return Ok(None); - } - let inputs = self.read_bundle_inputs(&overlay.bundle, true)?; - if !inputs.independent || inputs.target != overlay.final_position { - return Err(LtxError::ChecksumMismatch); - } - self.shared_inputs(inputs.inputs, inputs.target) - .await - .map(Some) - } - - async fn shared_inputs( - &self, - inputs: Vec, - position: Position, - ) -> Result { let segments: Vec<_> = inputs .into_iter() .map(|input| PreparedSegment { @@ -229,13 +165,13 @@ impl CellReplica { .and_then(|bytes| bytes.checked_add(segment.descriptor.index_length)) .ok_or(LtxError::LTXCorrupted) })?; - Ok(SharedCaptures { + Ok(Some(SharedCaptures { replica: self.clone(), - position, + position: cuts.position, segments, original, encoded_bytes, - }) + })) } /// Uploads a multi-row cohort once and returns independently scoped inputs. diff --git a/crates/cellule-ltx/tests/cell/roots.rs b/crates/cellule-ltx/tests/cell/roots.rs index 908abe22..97d15505 100644 --- a/crates/cellule-ltx/tests/cell/roots.rs +++ b/crates/cellule-ltx/tests/cell/roots.rs @@ -42,5 +42,4 @@ mod packed; mod preparation; mod prepare_cost; mod shared; -mod shared_recovery; mod sparse; diff --git a/crates/cellule-ltx/tests/cell/roots/lifecycle.rs b/crates/cellule-ltx/tests/cell/roots/lifecycle.rs index 1334c0a4..bad94eb6 100644 --- a/crates/cellule-ltx/tests/cell/roots/lifecycle.rs +++ b/crates/cellule-ltx/tests/cell/roots/lifecycle.rs @@ -849,13 +849,6 @@ async fn recovered_overlay_keeps_the_bundle_when_a_later_row_exceeds_the_pack_bu ) .unwrap(); let overlay = RecoveryOverlay::new(base, bundle, last.position, 3); - assert!( - replica - .shared_recovered_captures(&overlay) - .await - .unwrap() - .is_none() - ); let recovered = replica .prepare_recovered_overlay(&overlay, 1) .await diff --git a/crates/cellule-ltx/tests/cell/roots/shared_recovery.rs b/crates/cellule-ltx/tests/cell/roots/shared_recovery.rs deleted file mode 100644 index f495725a..00000000 --- a/crates/cellule-ltx/tests/cell/roots/shared_recovery.rs +++ /dev/null @@ -1,176 +0,0 @@ -use super::*; - -fn bundle(cells: &[([u8; 32], [u8; 16], CaptureBatch)]) -> Bundle { - Bundle::encode( - cells - .iter() - .flat_map(|(cell, incarnation, cuts)| { - cuts.segments.iter().map(|segment| { - BundleEntry::for_cell( - *cell, - *incarnation, - segment.info().clone(), - std::fs::read(segment.path()).unwrap(), - ) - }) - }) - .collect(), - Limits::default(), - ) - .unwrap() -} - -#[tokio::test] -async fn shared_recovery_preserves_exact_scopes_original_chains_and_cold_bytes() { - let directory = tempfile::tempdir().unwrap(); - let counted = Arc::new(cellule_store::test_support::CountingObjectStore::new( - Arc::new(InMemory::new()), - )); - let store = Store::new(counted.clone()); - let mut replicas = Vec::new(); - let mut bases = Vec::new(); - let mut cells = Vec::new(); - for i in 0..2_u8 { - let cell = [91 + i; 32]; - let incarnation = [93 + i; 16]; - let replica = replica(store.clone(), cell, incarnation); - let mut db = Db::open( - &directory.path().join(format!("source-{i}")), - Limits::default(), - ) - .unwrap(); - db.transaction(|tx| tx.execute_batch("CREATE TABLE t(v); INSERT INTO t VALUES(1)")) - .unwrap(); - bases.push( - replica - .prepare(None, &db.capture().unwrap(), 1, 1) - .await - .unwrap() - .root(), - ); - db.transaction(|tx| tx.execute("INSERT INTO t VALUES(2)", [])) - .unwrap(); - let mut cuts = db.capture().unwrap(); - db.transaction(|tx| tx.execute("INSERT INTO t VALUES(3)", [])) - .unwrap(); - let last = db.capture().unwrap(); - cuts.segments.extend(last.segments); - cuts.position = last.position; - cells.push((cell, incarnation, cuts)); - replicas.push(replica); - db.close().unwrap(); - } - let mut captures = Vec::new(); - for (i, replica) in replicas.iter().enumerate() { - let overlay = RecoveryOverlay::new(bases[i], bundle(&cells), cells[i].2.position, 3); - let (_, rows) = overlay.shared_input_upper_bound().unwrap(); - assert_eq!(rows, 2, "only this Cell's original rows are charged"); - captures.push( - replica - .shared_recovered_captures(&overlay) - .await - .unwrap() - .unwrap(), - ); - } - counted.reset(); - let appends = CellReplica::upload_shared(captures, directory.path()) - .await - .unwrap(); - assert_eq!(counted.put_requests(), 1, "one shared immutable body"); - assert!( - replicas[1] - .prepare_shared(Some(&bases[1]), &appends[0], 3, 1) - .await - .is_err() - ); - for (i, replica) in replicas.iter().enumerate() { - let shared = replica - .prepare_shared(Some(&bases[i]), &appends[i], 3, 1) - .await - .unwrap(); - let overlay = RecoveryOverlay::new(bases[i], bundle(&cells), cells[i].2.position, 3); - let canonical = replica - .prepare_recovered_overlay(&overlay, 1) - .await - .unwrap(); - assert_eq!(shared.predecessor(), Some(bases[i])); - assert_eq!(shared.root().position, canonical.root().position); - assert_eq!(shared.root().commit_sequence, 3); - let cold = super::replica(store.clone(), cells[i].0, cells[i].1); - let shared_path = directory.path().join(format!("shared-{i}")); - let direct_path = directory.path().join(format!("direct-{i}")); - cold.open_root(&shared.root()) - .await - .unwrap() - .restore(&shared_path) - .await - .unwrap(); - cold.open_root(&canonical.root()) - .await - .unwrap() - .restore(&direct_path) - .await - .unwrap(); - assert_eq!( - std::fs::read(shared_path).unwrap(), - std::fs::read(direct_path).unwrap() - ); - - let limited = CellReplica::new( - CellStorageLayout::new(store.clone(), Path::from("runtime"), [3; 16]), - cells[i].0, - cells[i].1, - Limits { - max_segments: 2, - ..Limits::default() - }, - ) - .unwrap(); - counted.reset(); - assert!( - limited - .prepare_shared(Some(&bases[i]), &appends[i], 3, 1) - .await - .is_err(), - "one base plus two original cuts exceeds admission despite coalescing" - ); - assert_eq!(counted.put_requests(), 0); - } -} - -#[tokio::test] -async fn shared_recovery_rejects_overlay_scope_and_endpoint_before_upload() { - let directory = tempfile::tempdir().unwrap(); - let counted = Arc::new(cellule_store::test_support::CountingObjectStore::new( - Arc::new(InMemory::new()), - )); - let replica = replica(Store::new(counted.clone()), [97; 32], [98; 16]); - let mut db = Db::open(&directory.path().join("source"), Limits::default()).unwrap(); - db.transaction(|tx| tx.execute_batch("CREATE TABLE t(v); INSERT INTO t VALUES(1)")) - .unwrap(); - let base = replica - .prepare(None, &db.capture().unwrap(), 1, 1) - .await - .unwrap() - .root(); - db.transaction(|tx| tx.execute("INSERT INTO t VALUES(2)", [])) - .unwrap(); - let cells = vec![([97; 32], [98; 16], db.capture().unwrap())]; - for (predecessor, endpoint) in [ - ( - RootRef { - cell: [99; 32], - ..base - }, - cells[0].2.position, - ), - (base, base.position), - ] { - counted.reset(); - let overlay = RecoveryOverlay::new(predecessor, bundle(&cells), endpoint, 2); - assert!(replica.shared_recovered_captures(&overlay).await.is_err()); - assert_eq!(counted.put_requests(), 0); - } - db.close().unwrap(); -} diff --git a/crates/cellule-runtime/docs/write-performance-design.md b/crates/cellule-runtime/docs/write-performance-design.md index 03392a2b..9127df86 100644 --- a/crates/cellule-runtime/docs/write-performance-design.md +++ b/crates/cellule-runtime/docs/write-performance-design.md @@ -133,15 +133,6 @@ selection cohorts serially. Publication staging, reduced repeated verification and complete application performance qualification remain required. Component passes establish no new TPS claim. -Selected bundle materialization now submits eligible recovered rows to the same -shared root producer used by native captures. The canonical reader verifies -original scoped rows before coalescing; the root factory still checks the fresh -predecessor, complete original chain, lineage and exact endpoint before Cell CAS. -Pre-admitted memory follows dispatched native jobs through cancellation. The -original 256-KiB/64-row bounds remain; large tails or retained-memory pressure -use direct recovery preparation. End-to-end TPS and total GET/PUT bytes must -establish whether this integration improves application performance. - ```mermaid flowchart LR A[Bounded admission before SQL] --> B[Mutation and retry result commit together] diff --git a/crates/cellule-runtime/src/node/bundle/tests/materialization_shared.rs b/crates/cellule-runtime/src/node/bundle/tests/materialization_shared.rs deleted file mode 100644 index 8ee77d60..00000000 --- a/crates/cellule-runtime/src/node/bundle/tests/materialization_shared.rs +++ /dev/null @@ -1,180 +0,0 @@ -use super::*; -use crate::fleet::resource::{ResourceCost, ResourceLedger}; -use crate::fleet::telemetry::{CellTelemetry, CellTelemetryHandle, SharedPublicationTiming}; -use crate::publication::SharedPublication; -use futures_util::{StreamExt as _, stream::FuturesUnordered}; -use std::sync::atomic::{AtomicU64, Ordering}; - -#[derive(Default)] -struct SharedEvidence { - cells: AtomicU64, - pressure: AtomicU64, -} - -impl CellTelemetry for SharedEvidence { - fn shared_publication(&self, timing: SharedPublicationTiming) { - assert!(timing.succeeded); - self.cells.fetch_add(timing.cells, Ordering::SeqCst); - } - - fn shared_publication_singleton(&self, _: std::time::Duration) { - self.cells.fetch_add(1, Ordering::SeqCst); - } - - fn shared_publication_fallback(&self, pressure: bool) { - assert!( - pressure, - "these small verified overlays fit the original byte bound" - ); - self.pressure.fetch_add(1, Ordering::SeqCst); - } -} - -#[tokio::test] -async fn selected_bundle_materialization_enters_shared_producer_and_recovers_exactly() { - materialize(8 << 20, 8, 0).await; -} - -#[tokio::test] -async fn selected_bundle_materialization_falls_back_under_original_memory_pressure() { - materialize(1, 0, 8).await; -} - -async fn materialize(memory: usize, shared_cells: u64, pressure: u64) { - let mut f = Fixture::new().await; - let host = cellule_ltx::Host::default() - .with_local_disk_budget(cellule_ltx::DiskBudget::new(8 << 20)) - .with_io_slots(Arc::new(tokio::sync::Semaphore::new(1))) - .with_job_slots(Arc::new(tokio::sync::Semaphore::new(1))) - .with_dirty_slots(Arc::new(tokio::sync::Semaphore::new(1))) - .with_recovery_slots(Arc::new(tokio::sync::Semaphore::new(1))) - .with_scratch_slots(Arc::new(tokio::sync::Semaphore::new(1))); - let disk = host.local_disk_budget(); - let ledger = ResourceLedger::new( - ResourceCost::zero() - .with_retained_bytes(memory) - .with_publication_file_descriptors(64), - ); - let evidence = Arc::new(SharedEvidence::default()); - let coordinator = SharedPublication::new( - ledger.clone(), - CellTelemetryHandle::from_sink(evidence.clone()), - ); - let mut cells = Vec::new(); - let mut frames = Vec::new(); - let mut assignments = Vec::new(); - for byte in 4..12 { - let mut cell = f.cell(byte).await; - cell.replica = cell.replica.with_host(host.clone()); - let (_, next, assigned) = f.append(&mut cell, 2); - frames.extend(next); - assignments.push(assigned); - cells.push(cell); - } - let prepared = f - .directory - .prepare_node_bundle(&f.node, &frames, &assignments, NOW) - .await - .unwrap(); - let (node, proofs) = f - .directory - .select_node_bundle(&f.node, &prepared, &f.lease, Limits::default(), NOW) - .await - .unwrap(); - f.node = node; - let mut tasks = FuturesUnordered::new(); - for cell in &cells { - let proof = proofs - .iter() - .find(|proof| proof.binding.control.cell == cell.control.value().cell) - .unwrap(); - let mut publisher = f - .publisher(cell) - .with_shared_publication(coordinator.clone()); - let expected_cell = cell.control.value().cell; - tasks.push(async move { - let root = publisher.materialize_bundle(proof).await.unwrap(); - (expected_cell, root) - }); - } - let mut roots = Vec::new(); - while let Some(result) = tokio::time::timeout(std::time::Duration::from_secs(5), tasks.next()) - .await - .unwrap() - { - roots.push(result); - } - assert_eq!(roots.len(), 8); - coordinator.shutdown().await.unwrap(); - assert_eq!(ledger.snapshot().unwrap().used, ResourceCost::zero()); - assert_eq!(disk.used(), 0); - for (cell, root) in roots { - assert_eq!(root.commit_sequence, 2); - let original = cells - .iter() - .find(|candidate| candidate.control.value().cell == cell) - .unwrap(); - let proof = proofs - .iter() - .find(|proof| proof.binding.control.cell == cell) - .unwrap(); - let overlay = proof - .recovery_overlay(original.authority.layout(), Limits::default()) - .await - .unwrap(); - let direct = original - .replica - .prepare_recovered_overlay(&overlay, 1) - .await - .unwrap(); - let cold = CellReplica::new( - original.authority.layout().clone(), - root.cell, - root.incarnation, - Limits::default(), - ) - .unwrap(); - cold.reachable_objects(&root).await.unwrap(); - let destination = f - .scratch - .path() - .join(format!("shared-{}", cell.as_bytes()[0])); - let canonical = f - .scratch - .path() - .join(format!("direct-{}", cell.as_bytes()[0])); - cold.open_root(&root) - .await - .unwrap() - .restore(&destination) - .await - .unwrap(); - cold.open_root(&direct.root()) - .await - .unwrap() - .restore(&canonical) - .await - .unwrap(); - assert_eq!( - std::fs::read(&destination).unwrap(), - std::fs::read(canonical).unwrap() - ); - let db = rusqlite::Connection::open(destination).unwrap(); - let outcomes: Vec<(String, String)> = db - .prepare("SELECT request,result FROM outcomes ORDER BY request") - .unwrap() - .query_map([], |row| Ok((row.get(0)?, row.get(1)?))) - .unwrap() - .collect::>() - .unwrap(); - assert_eq!( - outcomes, - [ - ("request-2".into(), "result-2".into()), - ("seed".into(), "original".into()) - ] - ); - } - assert_eq!(evidence.cells.load(Ordering::SeqCst), shared_cells); - assert_eq!(evidence.pressure.load(Ordering::SeqCst), pressure); -} diff --git a/crates/cellule-runtime/src/node/bundle/tests/mod.rs b/crates/cellule-runtime/src/node/bundle/tests/mod.rs index 68ed64db..ecd872ce 100644 --- a/crates/cellule-runtime/src/node/bundle/tests/mod.rs +++ b/crates/cellule-runtime/src/node/bundle/tests/mod.rs @@ -21,7 +21,6 @@ mod faults; mod index; mod lifecycle; mod managed; -mod materialization_shared; mod ranges; mod readiness; mod receipt_pressure; diff --git a/crates/cellule-runtime/src/publication/mod.rs b/crates/cellule-runtime/src/publication/mod.rs index b990d5fb..ff412d0b 100644 --- a/crates/cellule-runtime/src/publication/mod.rs +++ b/crates/cellule-runtime/src/publication/mod.rs @@ -11,7 +11,6 @@ use crate::retry::{Backoff, retry_hint, retryable_storage_error}; use crate::{Error, Result}; pub(crate) mod lineage; -mod recovered; mod shared; pub(crate) use shared::{PublicationPermit, SharedPublication}; @@ -130,7 +129,14 @@ impl CellPublisher { .recovery_overlay_from(self.authority.layout(), self.replica.limits(), base) .await?; self.check_node_lease()?; - let prepared = self.prepare_recovered(&overlay).await?; + let (replica, confirmation) = lineage::replica(self.replica.clone(), &self.authority); + let prepared = replica + .prepare_recovered_overlay(&overlay, self.observed.value().schema) + .await + .map_err(lineage::error)?; + self.lineage_confirmed = *confirmation + .lock() + .map_err(|_| Error::Peer("root lineage confirmation lock poisoned"))?; self.publish_prepared(&prepared, next_due_ms).await } diff --git a/crates/cellule-runtime/src/publication/recovered.rs b/crates/cellule-runtime/src/publication/recovered.rs deleted file mode 100644 index a135c292..00000000 --- a/crates/cellule-runtime/src/publication/recovered.rs +++ /dev/null @@ -1,88 +0,0 @@ -//! Selected overlays enter the same bounded producer and exact root factory. -use super::*; - -impl CellPublisher { - pub(super) async fn prepare_recovered( - &mut self, - overlay: &cellule_ltx::RecoveryOverlay, - ) -> Result { - let result = match self.admit_publication().await? { - PublicationPermit::Direct(replica) => { - self.prepare_recovered_inputs(*replica, overlay, None).await - } - PublicationPermit::Shared(slot) => { - let coordinator = self - .shared_publication - .clone() - .ok_or(Error::Control("shared publication is unavailable"))?; - let replica = self.replica.clone(); - let submission = coordinator.submit_recovered( - &replica, - overlay, - self.scratch_directory.clone(), - slot, - ); - tokio::pin!(submission); - let shared = loop { - tokio::select! { - result = &mut submission => break result, - _ = tokio::time::sleep_until(tokio::time::Instant::from_std(self.renew_at)) => self.renew().await?, - } - }; - self.check_node_lease()?; - let shared = match shared { - Ok(shared) => shared, - // A failed sibling/upload cannot invalidate this Cell's - // original overlay. The direct factory rechecks its exact - // inputs and retains its own source error on failure. - Err(Error::Shared(source)) if matches!(source.as_ref(), Error::Ltx(_)) => None, - Err(error) => return Err(error), - }; - self.prepare_recovered_inputs(self.replica.clone(), overlay, shared.as_ref()) - .await - } - }; - self.record_publication_cost(); - result - } - - async fn prepare_recovered_inputs( - &mut self, - replica: cellule_ltx::CellReplica, - overlay: &cellule_ltx::RecoveryOverlay, - shared: Option<&shared::SharedPrepared>, - ) -> Result { - self.check_node_lease()?; - let (replica, confirmation) = lineage::replica(replica, &self.authority); - let base = overlay.predecessor(); - let schema = self.observed.value().schema; - let preparation = match shared { - Some(shared) => futures_util::future::Either::Left(replica.prepare_shared( - Some(&base), - &shared.append, - overlay.final_commit_sequence(), - schema, - )), - None => futures_util::future::Either::Right( - replica.prepare_recovered_overlay(overlay, schema), - ), - }; - tokio::pin!(preparation); - let prepared = loop { - tokio::select! { - result = &mut preparation => break result.map_err(lineage::error)?, - _ = tokio::time::sleep_until(tokio::time::Instant::from_std(self.renew_at)) => self.renew().await?, - } - }; - if prepared.predecessor() != Some(base) - || prepared.root().position != overlay.final_position() - || prepared.root().commit_sequence != overlay.final_commit_sequence() - { - return Err(cellule_ltx::LtxError::ChecksumMismatch.into()); - } - self.lineage_confirmed = *confirmation - .lock() - .map_err(|_| Error::Peer("root lineage confirmation lock poisoned"))?; - Ok(prepared) - } -} diff --git a/crates/cellule-runtime/src/publication/shared/mod.rs b/crates/cellule-runtime/src/publication/shared/mod.rs index 7547943d..84cc82e0 100644 --- a/crates/cellule-runtime/src/publication/shared/mod.rs +++ b/crates/cellule-runtime/src/publication/shared/mod.rs @@ -30,20 +30,14 @@ pub(crate) enum PublicationPermit { pub(crate) struct SharedPrepared { pub(crate) append: SharedAppend, // Index/row tables remain charged through per-Cell root preparation. - _memory: Arc, + _memory: ResourceReservation, } -struct SharedMemory { - _reservation: ResourceReservation, -} - -impl cellule_ltx::HostResourcePermit for SharedMemory {} - struct Entry { captures: SharedCaptures, scratch: PathBuf, accepted_at: Instant, - memory: Arc, + memory: ResourceReservation, _slot: OwnedSemaphorePermit, reply: oneshot::Sender>, } @@ -106,60 +100,15 @@ impl SharedPublication { .checked_add(row) .ok_or(Error::Capacity("shared publication bytes")) })?; - let Some(memory) = - self.reserve_input(estimate, cuts.segments.len(), cuts.segments.len() + 2)? - else { - return Ok(None); - }; - let replica = replica.clone().with_preparation_resource(memory.clone()); - let Some(captures) = replica.shared_captures(cuts).await? else { - self.telemetry.shared_publication_fallback(false); - return Ok(None); - }; - self.submit_captures(captures, memory, scratch, slot) - .await - .map(Some) - } - - pub(crate) async fn submit_recovered( - &self, - replica: &cellule_ltx::CellReplica, - overlay: &cellule_ltx::RecoveryOverlay, - scratch: PathBuf, - slot: OwnedSemaphorePermit, - ) -> Result> { - let (estimate, rows) = overlay.shared_input_upper_bound()?; - let estimate = - usize::try_from(estimate).map_err(|_| Error::Capacity("shared publication bytes"))?; - // The enclosing overlay owns its original artifact. Verified frozen - // rows need only the bounded shared scratch descriptors, not one native - // file pin per row. Never retain a Cell dirty slot while enqueued. - let Some(memory) = self.reserve_input(estimate, rows, 2)? else { - return Ok(None); - }; - let replica = replica.clone().with_preparation_resource(memory.clone()); - let Some(captures) = replica.shared_recovered_captures(overlay).await? else { - self.telemetry.shared_publication_fallback(false); - return Ok(None); - }; - self.submit_captures(captures, memory, scratch, slot) - .await - .map(Some) - } - - fn reserve_input( - &self, - estimate: usize, - rows: usize, - file_descriptors: usize, - ) -> Result>> { - if estimate as u64 > SHARED_PUBLICATION_BYTES || rows > SHARED_PUBLICATION_ROWS { + if estimate as u64 > SHARED_PUBLICATION_BYTES + || cuts.segments.len() > SHARED_PUBLICATION_ROWS + { self.telemetry.shared_publication_fallback(false); return Ok(None); } // Charge before pinning/indexing/coalescing. Multiple compressed cuts // can expand into a bounded 256 KiB page map before being re-encoded. - let work = if rows > 1 { + let work = if cuts.segments.len() > 1 { estimate.max(SHARED_PUBLICATION_BYTES as usize) } else { estimate @@ -170,7 +119,7 @@ impl SharedPublication { .ok_or(Error::Capacity("shared publication memory"))?; let cost = ResourceCost::zero() .with_retained_bytes(memory) - .with_publication_file_descriptors(file_descriptors); + .with_publication_file_descriptors(cuts.segments.len() + 2); let memory = match self.resources.try_reserve(cost) { Ok(memory) => memory, // Sharing is optional representation reduction. Never wait for @@ -181,18 +130,10 @@ impl SharedPublication { } Err(error) => return Err(error), }; - Ok(Some(Arc::new(SharedMemory { - _reservation: memory, - }))) - } - - async fn submit_captures( - &self, - captures: SharedCaptures, - memory: Arc, - scratch: PathBuf, - slot: OwnedSemaphorePermit, - ) -> Result { + let Some(captures) = replica.shared_captures(cuts).await? else { + self.telemetry.shared_publication_fallback(false); + return Ok(None); + }; let (reply, response) = oneshot::channel(); self.sender .send(Message::Capture(Box::new(Entry { @@ -206,7 +147,7 @@ impl SharedPublication { .await .map_err(|_| Error::RuntimeClosed)?; // The lane, rather than the waiter, owns dispatched storage and scratch. - response.await.map_err(|_| Error::RuntimeClosed)? + response.await.map_err(|_| Error::RuntimeClosed)?.map(Some) } pub(crate) async fn shutdown(&self) -> Result<()> { diff --git a/crates/cellule-runtime/src/publication/shared/tests.rs b/crates/cellule-runtime/src/publication/shared/tests.rs index 69c09d80..0f127244 100644 --- a/crates/cellule-runtime/src/publication/shared/tests.rs +++ b/crates/cellule-runtime/src/publication/shared/tests.rs @@ -114,9 +114,7 @@ async fn minimum_host_permits_drain_shared_work_after_a_sibling_waiter_cancels() captures, scratch: directory.path().to_owned(), accepted_at: Instant::now(), - memory: Arc::new(SharedMemory { - _reservation: memory, - }), + memory, _slot: slot, reply, }); From a4ad3401e4d398363c54d000a279d010ddba3abc Mon Sep 17 00:00:00 2001 From: forhappy Date: Fri, 9 Oct 2026 17:16:00 -0700 Subject: [PATCH 098/102] Record rejected shared-overlay experiment and measured publication cost --- .../docs/write-performance-design.md | 13 ++ docs/pr67-shared-overlay-measurement.md | 126 ++++++++++++++++++ 2 files changed, 139 insertions(+) create mode 100644 docs/pr67-shared-overlay-measurement.md diff --git a/crates/cellule-runtime/docs/write-performance-design.md b/crates/cellule-runtime/docs/write-performance-design.md index 9127df86..e4f84b57 100644 --- a/crates/cellule-runtime/docs/write-performance-design.md +++ b/crates/cellule-runtime/docs/write-performance-design.md @@ -1,5 +1,18 @@ # Node write and read performance design +The latest [shared-overlay diagnostic](../../../docs/pr67-shared-overlay-measurement.md) +rejects and reverts recovered-overlay integration through the shared producer. +Fresh Fleet throughput falls 537.38→501.40/s and successful scheduled p99 rises +371.96→383.70 ms; fresh celld completes 1,999.78/s with 14.52-ms p99. Only 58 of +1,450 admitted Cells share an object; 1,005 preparations fall back under memory +pressure. Total observed store bytes/success rise 74,318→78,018, with node authority +accounting for about four-fifths. Both Cellule warm ACK audits fail, cold audit is +not reached, and debt grows. Prioritize composing ready checkpoints with native +selection under one bounded canonical catalog verification/CAS, better admission +and cohort density using original credits, and warm read/retry availability. +Production is restored; all profiles remain unqualified and acceptance gates +below are unchanged. Shared root packing alone does not deliver parity. + The latest [checkpoint-cohort diagnostic](../../../docs/pr67-checkpoint-cohort-measurement.md) rejects and reverts bounded parallel checkpoint verification. The controlled fixture overlaps eight fresh roots rather than one, but application TPS falls diff --git a/docs/pr67-shared-overlay-measurement.md b/docs/pr67-shared-overlay-measurement.md new file mode 100644 index 00000000..29f272b7 --- /dev/null +++ b/docs/pr67-shared-overlay-measurement.md @@ -0,0 +1,126 @@ +# PR 67: recovered-overlay shared publication experiment + +The shared-producer integration is **rejected and reverted**. Fresh matched Fleet +throughput falls **537.38→501.40 writes/s**; successful scheduled p99 rises +**371.96→383.70 ms**. Shared publication activates, but only 58 of 1,450 admitted +Cells share an object and 1,005 preparations fall back under memory pressure. +Total observed object-store bytes per successful write rise **74,318→78,018**. +Both Cellule arms fail warm availability checks. This establishes no performance +parity; PR #67 remains a draft. + +## Source and implementation + +| Role | Revision | +| --- | --- | +| Fresh retained Cellule binary | `cf4785c675698afe786f566242d6d3dace32e775` | +| Current baseline | `9f7738a94770910f896d1d61bd3bed2abcda6372`, identical Rust/Cargo production source | +| Rejected prototype | `c2c0ef10b08b9844401ca9e11397e9321a744f91` | +| Reversion | `ce3e8d6`, restores the complete baseline crate/Cargo tree | +| Fresh celld | `f2bf648663a610eefde71f3547ad61e9b896b1f0`, pinned image | + +The prototype extracts the canonical recovered-bundle input reader, routes +eligible materialization through the existing shared producer, retains original +chain facts before coalescing and preserves lineage and exact endpoint checks +before fenced Cell CAS. A scoped host token retains pre-admitted working memory +through cancelled native input preparation. Large or memory-constrained inputs retain +direct recovery preparation. Original object, row, memory and scheduling bounds +remain unchanged. No copied celld code, persisted format change or weaker proof +is delivered. Rejected source remains in Git history and external evidence. + +The comparison remains celld's [ordered shipping](https://github.com/denoland/celld/blob/f2bf648663a610eefde71f3547ad61e9b896b1f0/crates/celld/ltx_repl.rs#L4530), +[follower group commit](https://github.com/denoland/celld/blob/f2bf648663a610eefde71f3547ad61e9b896b1f0/crates/celld/node_log.rs#L179) +and [new-entry publication](https://github.com/denoland/celld/blob/f2bf648663a610eefde71f3547ad61e9b896b1f0/crates/celld/node_log.rs#L6628). +Cellule retains the preceding admission-before-ordering, eight ordered follower +rounds, fresh-origin encoder reuse and completed-command credit transfer. + +## Fresh application measurement + +One owner, two followers, 2,000 uniform active Cells, 96-byte SQL-ledger values, +WAL NORMAL/tmpfs, 128 clients and queue slots, 2,000 offered writes/s, 30-second +warmup and 60-second measured window. Driver and auditor binaries are byte +identical; workload, fixtures and pinned images match. The arms run sequentially. +Builds, tests and replay do not overlap timed windows; original controllers join +before independent journal analysis. + +| Arm | Successful writes/s | Successful scheduled p99 ms | Errors | Dropped offers | +| --- | ---: | ---: | ---: | ---: | +| Retained Cellule | 537.38 | 371.96 | 61,556 | 26,201 | +| Rejected prototype | 501.40 | 383.70 | 51,127 | 38,789 | +| Fresh celld | 1,999.78 | 14.52 | 0 | 0 | + +TPS counts completions inside the measured window. P99 covers successful +measured offers through client drain and excludes failed requests. Successful +request p99 is 201.93/214.82/13.36 ms; all-attempt scheduled p99 is +328.47/389.81/14.52 ms. Cellule has no trailing successful completions; celld has +13, excluded from TPS. All 2,000 Cells have measured successes in every arm. +Warmup errors/drops are 28,113/6,132, 26,104/7,602 and 0/1,602 respectively. +One short sequential pair establishes no repeatable causal attribution. + +Independent full-journal replay reconciles every offer, counter, original output, +payload, per-Cell count and ACK provenance. Retained Cellule's 59,999-ACK warm +audit fails 1,152 HTTP 503 checks; the prototype's 58,379-ACK warm audit fails 45. +Neither reaches cold audit. Both log all 2,000 Cells drained to Idle and all three +processes exit zero during cleanup; neither logs an owner fence or an unknown +write outcome in this run. Cleanup success does not replace a passing audit. +Celld passes all 180,399 warm/cold mutations and original retries and drains in +12.29 seconds. These facts do not establish lost acknowledged data or a +qualified failure/transfer/collection matrix. + +## Observed work and remaining architecture gap + +| Measured-window metric | Retained | Prototype | +| --- | ---: | ---: | +| Shared root cohorts / Cells | 0 / 0 | 29 / 58 | +| Shared singletons / pressure fallbacks | 0 / 0 | 1,392 / 1,005 | +| Immutable GET bytes | 433,418,483 | 385,574,161 | +| Immutable PUT bytes | 27,035,528 | 25,253,095 | +| Node-authority range bytes | 1,113,485,391 | 1,160,104,166 | +| Node-authority PUT bytes | 428,672,754 | 405,013,158 | +| Node-authority total bytes/success | 59,837 | 64,163 | +| All store read+write bytes/success | 74,318 | 78,018 | +| Native publication debt bytes, start→end | 30,614,543→46,599,220 | 35,265,911→43,058,711 | +| Runtime retained bytes, start→end | 49,099,899→64,629,624 | 66,781,638→64,956,731 | + +Storage totals include every observed operation in each family. These are window +cost ratios, not isolated operation costs. Node-authority traffic accounts for +about four-fifths of total observed store bytes. Only 4% of admitted shared Cells +actually share an object; reducing the small immutable-upload component does +not remove catalog/history and root-checkpoint work. Debt grows in both arms. + +Mean ordered-lock waits are 0.00085/0.00056 ms. Follower batches average +11.80–11.82/10.70–10.71 input frames per sync, with zero follower append failures. +Fleet proof waits average 52.84/62.28 ms in separate populations. The retained +submission phase populations differ by one at the sample boundary, so no exact +closed partition is claimed for that arm; the prototype's counts and nanoseconds +reconcile. The failed first analysis and corrected interpretation are preserved. + +Next work must reduce authenticated catalog and checkpoint work per command: +compose ready root checkpoints with native selection under one bounded fresh +catalog verification and fenced CAS, preserve exact original range/lineage +proofs, and improve materializer admission and cohort density using existing +credits. Avoid repeated historical reads within the same verified operation; +retain fresh dependency verification and complete issued-suffix recovery. Fix +warm read/retry availability under backlog and measure each candidate before +retaining it. Shared packing alone has not delivered the required architecture +or sustainable throughput. + +## Verification and limits + +The baseline regression reproduces zero shared-producer entries after exact cold +recovery. Three candidate repetitions pass all eight Cells and memory-pressure +fallback; public LTX tests enforce scope, endpoint, original-chain admission and +byte-identical cold restore. The host cancellation test, all 103 bundle tests, +shared-publication tests and all 13 isolated contributor routes pass. Workspace +logs report 1,998 test/doctest executions including child-process reporting, +38 environment ignores and 60 local LTX tests. Invalid fixture and compilation +attempts are preserved separately. Reversion restores previously verified +production byte-identically; final document checks cover the delivered report. + +All canonical qualification reports are false. The shared 8-CPU/8.3-GB Docker VM +does not qualify a dedicated 8-vCPU/16-GiB owner; tmpfs does not qualify +physical-media durability. Cellule mTLS/signed protobuf and celld loopback HTTP +differ. New Bucket/read/mixed measurements, three paired ≥5-minute repetitions, +zero errors/drops, tail targets, complete recovery/transfer/collection evidence +and stable debt remain required. No qualification profiles, deadlines, budgets or +expected assertions are weakened. Raw evidence stays outside Git at +`/Volumes/Workspace/crabbuild-target/native-shared-overlay-20261009`. From 23f309d4e56eb80c1a5e4eb2cf4ea6a279df3dd5 Mon Sep 17 00:00:00 2001 From: forhappy Date: Fri, 9 Oct 2026 17:41:05 -0700 Subject: [PATCH 099/102] Compose ready root checkpoints with native bundle publication --- .../cellule-axum/examples/fleet/authority.rs | 63 ++- .../docs/write-performance-design.md | 7 + .../src/node/bundle/closure.rs | 129 ++++--- crates/cellule-runtime/src/node/bundle/mod.rs | 6 +- .../src/node/bundle/selection.rs | 44 ++- .../src/node/bundle/tests/actor.rs | 52 ++- .../src/node/bundle/tests/composed.rs | 179 +++++++++ .../src/node/bundle/tests/index/composed.rs | 358 ++++++++++++++++++ .../src/node/bundle/tests/index/mod.rs | 1 + .../src/node/bundle/tests/managed.rs | 1 + .../src/node/bundle/tests/mod.rs | 1 + .../src/node/bundle/tests/readiness.rs | 3 +- .../src/node/bundle/tests/receipt_pressure.rs | 5 +- .../durability/publication/checkpoints.rs | 63 +++ .../src/node/durability/publication/mod.rs | 97 +++-- 15 files changed, 863 insertions(+), 146 deletions(-) create mode 100644 crates/cellule-runtime/src/node/bundle/tests/composed.rs create mode 100644 crates/cellule-runtime/src/node/bundle/tests/index/composed.rs create mode 100644 crates/cellule-runtime/src/node/durability/publication/checkpoints.rs diff --git a/crates/cellule-axum/examples/fleet/authority.rs b/crates/cellule-axum/examples/fleet/authority.rs index a4f69b41..228b425f 100644 --- a/crates/cellule-axum/examples/fleet/authority.rs +++ b/crates/cellule-axum/examples/fleet/authority.rs @@ -268,6 +268,7 @@ impl NodeBundlePublicationAuthority for Authority { fn select<'a>( &'a self, captures: &'a [cellule_runtime::node::log_shipper::AssignedCapture], + checkpoints: &'a [BundleCheckpoint], lease: &'a NodeLeaseGuard, ) -> BoxFuture<'a, Result>> { Box::pin(async move { @@ -278,9 +279,17 @@ impl NodeBundlePublicationAuthority for Authority { .flat_map(|c| c.frames().iter().cloned()) .collect::>(); let assignments = captures.iter().map(|c| c.assignment()).collect::>(); + let ready = ready_checkpoints(checkpoints).await?; let prepared = self .directory - .prepare_node_bundle(¤t, &frames, &assignments, clock()?) + .prepare_node_bundle_with_checkpoints( + ¤t, + &frames, + &assignments, + &ready, + cellule_ltx::Limits::default(), + clock()?, + ) .await?; let (next, proofs) = self .directory @@ -302,26 +311,7 @@ impl NodeBundlePublicationAuthority for Authority { Box::pin(async move { let mut state = self.state.lock().await; let current = self.current(&state, 1).await?; - let mut ready = Vec::new(); - for checkpoint in checkpoints { - let root = checkpoint.root(); - let observed = checkpoint - .authority() - .load(cellule_runtime::CellId::from_bytes(root.cell)) - .await? - .ok_or(Error::Fenced)?; - let actual = observed.value().ltx_root().ok_or(Error::Fenced)?; - if observed.value().bundle_binding == Some(checkpoint.proof().binding()) - && actual.cell == root.cell - && actual.incarnation == root.incarnation - && actual.commit_sequence > root.commit_sequence - { - // The later root's original notification is retained by its - // joined publisher. This observation releases no locators. - continue; - } - ready.push((checkpoint.authority(), checkpoint.proof())); - } + let ready = ready_checkpoints(checkpoints).await?; state.observed = if ready.is_empty() { current } else { @@ -414,3 +404,34 @@ impl NodeLogAuthority for Authority { }) } } + +async fn ready_checkpoints( + checkpoints: &[BundleCheckpoint], +) -> Result< + Vec<( + &cellule_runtime::control::authority::CellAuthority, + &cellule_runtime::node::bundle::BundleCoverageProof, + )>, +> { + let mut ready = Vec::new(); + for checkpoint in checkpoints { + let root = checkpoint.root(); + let observed = checkpoint + .authority() + .load(cellule_runtime::CellId::from_bytes(root.cell)) + .await? + .ok_or(Error::Fenced)?; + let actual = observed.value().ltx_root().ok_or(Error::Fenced)?; + if observed.value().bundle_binding == Some(checkpoint.proof().binding()) + && actual.cell == root.cell + && actual.incarnation == root.incarnation + && actual.commit_sequence > root.commit_sequence + { + // The later root's original notification is retained by its + // joined publisher. This observation releases no locators. + continue; + } + ready.push((checkpoint.authority(), checkpoint.proof())); + } + Ok(ready) +} diff --git a/crates/cellule-runtime/docs/write-performance-design.md b/crates/cellule-runtime/docs/write-performance-design.md index e4f84b57..c5c27e13 100644 --- a/crates/cellule-runtime/docs/write-performance-design.md +++ b/crates/cellule-runtime/docs/write-performance-design.md @@ -1,5 +1,12 @@ # Node write and read performance design +The current candidate composes ready materialized checkpoints and new native +captures into one fresh catalog load, upload and fenced selection CAS. Exact +prefix proofs, original callbacks, the 64-row limit and the 20-MiB publication +working reservation remain in force. Idle and receipt-pressure paths still join +standalone checkpoints. Measurement and qualification determine whether this +candidate is retained; the diagnostic history below remains separate evidence. + The latest [shared-overlay diagnostic](../../../docs/pr67-shared-overlay-measurement.md) rejects and reverts recovered-overlay integration through the shared producer. Fresh Fleet throughput falls 537.38→501.40/s and successful scheduled p99 rises diff --git a/crates/cellule-runtime/src/node/bundle/closure.rs b/crates/cellule-runtime/src/node/bundle/closure.rs index 2fcfb408..e5b18a11 100644 --- a/crates/cellule-runtime/src/node/bundle/closure.rs +++ b/crates/cellule-runtime/src/node/bundle/closure.rs @@ -40,63 +40,11 @@ impl NodeDirectory { .advertisement .bundle .ok_or(Error::Node("bundle lane is absent"))?; - let mut pins = std::collections::HashSet::new(); - let mut cells = std::collections::BTreeSet::new(); - for (_, proof) in checkpoints { - if proof.pin.session != observed.advertisement.session - || proof.pin.epoch != head.epoch - || !pins.insert(proof.pin.digest) - { - return Err(Error::Fenced); - } - cells.insert(( - *proof.binding.application.as_bytes(), - *proof.binding.control.cell.as_bytes(), - )); - } + let cells = checkpoint_cells(observed, checkpoints)?; let mut catalog = store::load_catalog_cells(&self.layout, observed.advertisement.session, head, &cells) .await?; - let mut changed = false; - for (authority, proof) in checkpoints { - let pin = proof.binding(); - let binding = catalog.binding_mut(pin.digest)?; - if authority.layout().application_id() != binding.application.as_bytes() - || authority.layout().node_path(pin.session.as_bytes()) - != self.layout.node_path(pin.session.as_bytes()) - || authority.layout().immutable_cache_identity() - != self.layout.immutable_cache_identity() - { - return Err(Error::Fenced); - } - let current = authority - .load(binding.control.cell) - .await? - .ok_or(Error::Fenced)?; - let control = current.value(); - if control.bundle_binding != Some(pin) - || control.epoch != binding.control.epoch - || control.incarnation != binding.control.incarnation - || control.code != binding.control.code - || control.schema != binding.control.schema - { - return Err(Error::PendingPublication); - } - let root = control.ltx_root().ok_or(Error::PendingPublication)?; - let prefix = materialized_prefix(binding, proof, &root)?; - if prefix == 0 && Some(root) == binding.control.ltx_root() { - // Repeating an exact installed checkpoint is a no-op even if - // newer captures remain selected beyond this materialized base. - continue; - } - binding.control = control.clone(); - verify_base(&self.layout, binding, limits).await?; - binding.locators.drain(..prefix); - if binding.locators.is_empty() { - binding.first_commit = binding.selected_commit; - } - changed = true; - } + let changed = apply_checkpoints(&self.layout, &mut catalog, checkpoints, limits).await?; if !changed { return Ok(observed.clone()); } @@ -257,3 +205,76 @@ fn materialized_prefix( } Ok(locators.len()) } + +/// Shares exact checkpoint validation with native append preparation. +pub(super) fn checkpoint_cells( + observed: &VersionedNodeAdvertisement, + checkpoints: &[(&CellAuthority, &BundleCoverageProof)], +) -> Result> { + if checkpoints.len() > MAX_FRAMES { + return Err(Error::Capacity("bundle checkpoint count")); + } + let mut pins = std::collections::HashSet::new(); + let mut cells = std::collections::BTreeSet::new(); + for (_, proof) in checkpoints { + if proof.pin.session != observed.advertisement.session + || proof.pin.epoch != observed.advertisement.bundle.ok_or(Error::Fenced)?.epoch + || !pins.insert(proof.pin.digest) + { + return Err(Error::Fenced); + } + cells.insert(( + *proof.binding.application.as_bytes(), + *proof.binding.control.cell.as_bytes(), + )); + } + Ok(cells) +} + +pub(super) async fn apply_checkpoints( + layout: &cellule_ltx::CellStorageLayout, + catalog: &mut Catalog, + checkpoints: &[(&CellAuthority, &BundleCoverageProof)], + limits: cellule_ltx::Limits, +) -> Result { + let mut changed = false; + for (authority, proof) in checkpoints { + let pin = proof.binding(); + let binding = catalog.binding_mut(pin.digest)?; + if authority.layout().application_id() != binding.application.as_bytes() + || authority.layout().node_path(pin.session.as_bytes()) + != layout.node_path(pin.session.as_bytes()) + || authority.layout().immutable_cache_identity() != layout.immutable_cache_identity() + { + return Err(Error::Fenced); + } + let current = authority + .load(binding.control.cell) + .await? + .ok_or(Error::Fenced)?; + let control = current.value(); + if control.bundle_binding != Some(pin) + || control.epoch != binding.control.epoch + || control.incarnation != binding.control.incarnation + || control.code != binding.control.code + || control.schema != binding.control.schema + { + return Err(Error::PendingPublication); + } + let root = control.ltx_root().ok_or(Error::PendingPublication)?; + let prefix = materialized_prefix(binding, proof, &root)?; + if prefix == 0 && Some(root) == binding.control.ltx_root() { + // Repeating an exact installed checkpoint is a no-op even if + // newer captures remain selected beyond this materialized base. + continue; + } + binding.control = control.clone(); + verify_base(layout, binding, limits).await?; + binding.locators.drain(..prefix); + if binding.locators.is_empty() { + binding.first_commit = binding.selected_commit; + } + changed = true; + } + Ok(changed) +} diff --git a/crates/cellule-runtime/src/node/bundle/mod.rs b/crates/cellule-runtime/src/node/bundle/mod.rs index b475d889..cc0d0555 100644 --- a/crates/cellule-runtime/src/node/bundle/mod.rs +++ b/crates/cellule-runtime/src/node/bundle/mod.rs @@ -14,6 +14,7 @@ //! use cellule_runtime::node::{NodeDirectory, VersionedNodeAdvertisement}; //! use cellule_runtime::node::bundle::BundleCoverageProof; //! use cellule_runtime::node::lease::NodeLeaseGuard; +//! use cellule_runtime::control::authority::CellAuthority; //! use cellule_runtime::node::log::AssignedCommitRange; //! //! async fn select_complete_captures( @@ -22,10 +23,13 @@ //! lease: &NodeLeaseGuard, //! frames: &[cellule_ltx::VerifiedNodeFrame], //! assignments: &[AssignedCommitRange], +//! checkpoints: &[(&CellAuthority, &BundleCoverageProof)], //! now_ms: i64, //! ) -> cellule_runtime::Result<(VersionedNodeAdvertisement, Vec)> { //! let proposal = directory -//! .prepare_node_bundle(observed, frames, assignments, now_ms).await?; +//! .prepare_node_bundle_with_checkpoints( +//! observed, frames, assignments, checkpoints, cellule_ltx::Limits::default(), now_ms, +//! ).await?; //! directory.select_node_bundle( //! observed, &proposal, lease, cellule_ltx::Limits::default(), now_ms, //! ).await diff --git a/crates/cellule-runtime/src/node/bundle/selection.rs b/crates/cellule-runtime/src/node/bundle/selection.rs index e2e6a03c..ebffc410 100644 --- a/crates/cellule-runtime/src/node/bundle/selection.rs +++ b/crates/cellule-runtime/src/node/bundle/selection.rs @@ -101,25 +101,53 @@ impl NodeDirectory { frames: &[cellule_ltx::VerifiedNodeFrame], assignments: &[crate::node::log::AssignedCommitRange], now_ms: i64, + ) -> Result { + self.prepare_node_bundle_with_checkpoints( + observed, + frames, + assignments, + &[], + cellule_ltx::Limits::default(), + now_ms, + ) + .await + } + + /// Combines exact materialized checkpoint prefixes and new native captures + /// in one fresh catalog read and upload. Selection still verifies canonical + /// origin dependencies and advances the original fenced node CAS. + /// Frames plus checkpoint notifications retain the existing 64-row bound. + pub async fn prepare_node_bundle_with_checkpoints( + &self, + observed: &VersionedNodeAdvertisement, + frames: &[cellule_ltx::VerifiedNodeFrame], + assignments: &[crate::node::log::AssignedCommitRange], + checkpoints: &[( + &crate::control::authority::CellAuthority, + &BundleCoverageProof, + )], + limits: cellule_ltx::Limits, + now_ms: i64, ) -> Result { self.validate(&observed.advertisement, now_ms)?; let head = observed .advertisement .bundle .ok_or(Error::Node("bundle lane is absent"))?; - if frames.is_empty() || frames.len() > MAX_FRAMES { + if frames.is_empty() || frames.len() + checkpoints.len() > MAX_FRAMES { return Err(Error::Capacity("bundle frame count")); } - let cells = frames - .iter() - .map(|frame| { - let scope = frame.scope(); - (scope.application, scope.cell) - }) - .collect(); + let mut cells = closure::checkpoint_cells(observed, checkpoints)?; + cells.extend(frames.iter().map(|frame| { + let scope = frame.scope(); + (scope.application, scope.cell) + })); let mut catalog = store::load_catalog_cells(&self.layout, observed.advertisement.session, head, &cells) .await?; + // Release only the proven materialized prefix before extending any + // binding. Its later selected suffix and complete issued range survive. + closure::apply_checkpoints(&self.layout, &mut catalog, checkpoints, limits).await?; let mut consumed = 0_usize; for assignment in assignments { let count = usize::try_from( diff --git a/crates/cellule-runtime/src/node/bundle/tests/actor.rs b/crates/cellule-runtime/src/node/bundle/tests/actor.rs index b359dc07..44b0f384 100644 --- a/crates/cellule-runtime/src/node/bundle/tests/actor.rs +++ b/crates/cellule-runtime/src/node/bundle/tests/actor.rs @@ -109,6 +109,7 @@ impl NodeBundlePublicationAuthority for Authority { fn select<'a>( &'a self, captures: &'a [crate::node::log_shipper::AssignedCapture], + checkpoints: &'a [BundleCheckpoint], lease: &'a NodeLeaseGuard, ) -> BoxFuture<'a, Result>> { Box::pin(async move { @@ -118,9 +119,17 @@ impl NodeBundlePublicationAuthority for Authority { .flat_map(|c| c.frames().iter().cloned()) .collect::>(); let assignments = captures.iter().map(|c| c.assignment()).collect::>(); + let ready = ready_checkpoints(checkpoints).await?; let prepared = self .directory - .prepare_node_bundle(&node, &frames, &assignments, NOW) + .prepare_node_bundle_with_checkpoints( + &node, + &frames, + &assignments, + &ready, + Limits::default(), + NOW, + ) .await?; let (next, proofs) = self .directory @@ -134,23 +143,7 @@ impl NodeBundlePublicationAuthority for Authority { fn checkpoint<'a>(&'a self, checkpoints: &'a [BundleCheckpoint]) -> BoxFuture<'a, Result<()>> { Box::pin(async move { let mut node = self.observed.lock().await; - let mut ready = Vec::new(); - for checkpoint in checkpoints { - let current = checkpoint - .authority() - .load(CellId::from_bytes(checkpoint.root().cell)) - .await? - .ok_or(Error::Fenced)?; - let root = current.value().ltx_root().ok_or(Error::Fenced)?; - if current.value().bundle_binding == Some(checkpoint.proof().binding()) - && root.cell == checkpoint.root().cell - && root.incarnation == checkpoint.root().incarnation - && root.commit_sequence > checkpoint.root().commit_sequence - { - continue; - } - ready.push((checkpoint.authority(), checkpoint.proof())); - } + let ready = ready_checkpoints(checkpoints).await?; if !ready.is_empty() { *node = self .directory @@ -657,3 +650,26 @@ async fn actor_case(prior_fleet: bool, cancel_caller: bool, early_selection: boo 0 ); } + +async fn ready_checkpoints( + checkpoints: &[BundleCheckpoint], +) -> Result> { + let mut ready = Vec::new(); + for checkpoint in checkpoints { + let current = checkpoint + .authority() + .load(CellId::from_bytes(checkpoint.root().cell)) + .await? + .ok_or(Error::Fenced)?; + let root = current.value().ltx_root().ok_or(Error::Fenced)?; + if current.value().bundle_binding == Some(checkpoint.proof().binding()) + && root.cell == checkpoint.root().cell + && root.incarnation == checkpoint.root().incarnation + && root.commit_sequence > checkpoint.root().commit_sequence + { + continue; + } + ready.push((checkpoint.authority(), checkpoint.proof())); + } + Ok(ready) +} diff --git a/crates/cellule-runtime/src/node/bundle/tests/composed.rs b/crates/cellule-runtime/src/node/bundle/tests/composed.rs new file mode 100644 index 00000000..cbe13c0e --- /dev/null +++ b/crates/cellule-runtime/src/node/bundle/tests/composed.rs @@ -0,0 +1,179 @@ +//! Original producer callbacks join a combined checkpoint/native selection. +use super::actor::{Authority, transport}; +use super::receipt_pressure::submission; +use super::*; +use crate::cell::worker::SqlWorkerPool; +use crate::node::durability::{ + BundleCheckpoint, NodeBundleAuthority, NodeBundlePublicationAuthority, NodeDurability, +}; +use crate::node::log_shipper::{AssignedCapture, NodeLogShipper}; +use futures_util::{future::BoxFuture, poll}; +use std::sync::atomic::{AtomicUsize, Ordering}; + +struct HeldSelection { + original: Arc, + calls: AtomicUsize, + combined: AtomicUsize, + entered: tokio::sync::Notify, + resume: tokio::sync::Notify, +} +impl NodeBundleAuthority for HeldSelection { + fn bind<'a>( + &'a self, + authority: &'a CellAuthority, + observed: &'a VersionedControl, + ) -> BoxFuture<'a, Result> { + self.original.bind(authority, observed) + } + fn close<'a>( + &'a self, + authority: &'a CellAuthority, + observed: &'a VersionedControl, + issued: crate::node::log::CellIssuedRange, + ) -> BoxFuture<'a, Result<()>> { + self.original.close(authority, observed, issued) + } +} +impl NodeBundlePublicationAuthority for HeldSelection { + fn select<'a>( + &'a self, + captures: &'a [AssignedCapture], + checkpoints: &'a [BundleCheckpoint], + lease: &'a NodeLeaseGuard, + ) -> BoxFuture<'a, Result>> { + Box::pin(async move { + let result = self.original.select(captures, checkpoints, lease).await?; + self.combined.fetch_add(checkpoints.len(), Ordering::SeqCst); + if self.calls.fetch_add(1, Ordering::SeqCst) == 1 { + self.entered.notify_one(); + self.resume.notified().await; + } + Ok(result) + }) + } + fn checkpoint<'a>(&'a self, checkpoints: &'a [BundleCheckpoint]) -> BoxFuture<'a, Result<()>> { + self.original.checkpoint(checkpoints) + } +} + +#[tokio::test] +async fn producer_combines_ready_callback_with_queued_native_work_and_joins_complete_drain() { + let mut f = Fixture::new().await; + super::coverage::enroll(&mut f).await; + let mut cell = f.cell(4).await; + let original = Arc::new(Authority { + directory: f.directory.clone(), + observed: tokio::sync::Mutex::new(f.node.clone()), + }); + let peers = transport(&f, true); + let shipper = NodeLogShipper::new(f.gate.clone(), peers.clone(), Limits::default()).unwrap(); + let durability = Arc::new(NodeDurability::new( + f.gate.clone(), + shipper, + original.clone(), + peers, + f.lease.clone(), + )); + let pool = SqlWorkerPool::new(1, 1).unwrap(); + pool.configure_retained_capacity(32 << 20).unwrap(); + durability + .attach_selection_resources(pool.resource_ledger()) + .unwrap(); + let held = Arc::new(HeldSelection { + original, + calls: AtomicUsize::new(0), + combined: AtomicUsize::new(0), + entered: tokio::sync::Notify::new(), + resume: tokio::sync::Notify::new(), + }); + durability.start_bundle_publication(held.clone()).unwrap(); + let first = durability + .submit_capture(submission(&mut cell, 2)) + .await + .unwrap(); + let first_selected = first.selection.as_ref().unwrap().selected().await.unwrap(); + let root = f + .publisher(&cell) + .materialize_bundle(&first_selected.proof) + .await + .unwrap(); + let second = durability + .submit_capture(submission(&mut cell, 3)) + .await + .unwrap(); + tokio::time::timeout(std::time::Duration::from_secs(5), held.entered.notified()) + .await + .unwrap(); + let third = durability + .submit_capture(submission(&mut cell, 4)) + .await + .unwrap(); + let checkpoint = tokio::task::unconstrained(durability.checkpoint_materialized( + cell.authority.clone(), + root, + first_selected.clone(), + )); + tokio::pin!(checkpoint); + // Remove only this test poll's cooperative budget: the available bounded + // send cannot yield before accepting the notification. Pending then means + // the original callback is queued and awaiting the held producer's completion. + assert!(poll!(checkpoint.as_mut()).is_pending()); + held.resume.notify_one(); + let (checkpoint, second_selected, third_selected) = + tokio::time::timeout(std::time::Duration::from_secs(5), async { + tokio::join!( + checkpoint, + second.selection.as_ref().unwrap().selected(), + third.selection.as_ref().unwrap().selected() + ) + }) + .await + .unwrap(); + checkpoint.unwrap(); + second_selected.unwrap(); + let last = third_selected.unwrap(); + assert_eq!(held.combined.load(Ordering::SeqCst), 1); + assert_eq!(last.proof.base().unwrap(), root); + assert_eq!(last.proof.commit_sequence(), 4); + assert_eq!(last.proof.locator_count(), 2); + cell.control = cell + .authority + .load(cell.control.value().cell) + .await + .unwrap() + .unwrap(); + let final_root = f + .publisher(&cell) + .materialize_bundle(&last.proof) + .await + .unwrap(); + durability + .checkpoint_materialized(cell.authority.clone(), final_root, last.clone()) + .await + .unwrap(); + cell.control = cell + .authority + .load(cell.control.value().cell) + .await + .unwrap() + .unwrap(); + durability + .close_bundle_cell(&cell.authority, &cell.control) + .await + .unwrap(); + durability.shutdown().await.unwrap(); + drop(first); + drop(second); + drop(third); + drop(first_selected); + drop(last); + assert_eq!( + pool.resource_ledger() + .snapshot() + .unwrap() + .used + .retained_bytes(), + 0 + ); + pool.shutdown().await.unwrap(); +} diff --git a/crates/cellule-runtime/src/node/bundle/tests/index/composed.rs b/crates/cellule-runtime/src/node/bundle/tests/index/composed.rs new file mode 100644 index 00000000..63b0cc69 --- /dev/null +++ b/crates/cellule-runtime/src/node/bundle/tests/index/composed.rs @@ -0,0 +1,358 @@ +use super::*; + +#[tokio::test] +async fn ready_checkpoint_and_new_capture_share_one_catalog_upload_and_cas() { + let mut f = Fixture::new().await; + let mut cell = f.cell(4).await; + let (_, frames, assigned) = f.append(&mut cell, 2); + let proposal = f + .directory + .prepare_node_bundle(&f.node, &frames, &[assigned], NOW) + .await + .unwrap(); + let (node, mut proofs) = f + .directory + .select_node_bundle(&f.node, &proposal, &f.lease, Limits::default(), NOW) + .await + .unwrap(); + f.node = node; + let original = proofs.pop().unwrap(); + let root = f + .publisher(&cell) + .materialize_bundle(&original) + .await + .unwrap(); + let (cuts, frames, assigned) = f.append(&mut cell, 3); + f.count.reset(); + let proposal = f + .directory + .prepare_node_bundle_with_checkpoints( + &f.node, + &frames, + &[assigned], + &[(&cell.authority, &original)], + Limits::default(), + NOW, + ) + .await + .unwrap(); + let (node, mut proofs) = f + .directory + .select_node_bundle(&f.node, &proposal, &f.lease, Limits::default(), NOW) + .await + .unwrap(); + f.node = node; + assert_eq!( + f.count.put_requests(), + 2, + "one catalog upload and one node authority CAS for both obligations" + ); + let selected = proofs.pop().unwrap(); + assert_eq!(selected.base().unwrap(), root); + assert_eq!(selected.commit_sequence(), 3); + assert_eq!(selected.locator_count(), frames.len()); + let prefix = original.materialized_prefix(root, None).unwrap(); + selected + .continues_selected_prefix(&original, Some(&prefix)) + .unwrap(); + let expected = cell + .replica + .prepare(Some(&root), &cuts, 3, 1) + .await + .unwrap(); + let current = cell + .authority + .load(cell.control.value().cell) + .await + .unwrap() + .unwrap(); + let cold = f + .directory + .load_bundle_coverage(&cell.authority, ¤t, Limits::default()) + .await + .unwrap(); + let overlay = cold + .recovery_overlay(&f.layout, Limits::default()) + .await + .unwrap(); + let recovered = cell + .replica + .prepare_recovered_overlay(&overlay, 1) + .await + .unwrap(); + let actual_path = f.scratch.path().join("combined-actual.sqlite"); + let expected_path = f.scratch.path().join("combined-expected.sqlite"); + cell.replica + .open_root(&recovered.root()) + .await + .unwrap() + .restore(&actual_path) + .await + .unwrap(); + cell.replica + .open_root(&expected.root()) + .await + .unwrap() + .restore(&expected_path) + .await + .unwrap(); + assert_eq!( + std::fs::read(actual_path).unwrap(), + std::fs::read(expected_path).unwrap() + ); +} + +#[tokio::test] +async fn combined_cohort_preserves_independent_cells_and_a_prior_hot_suffix() { + let mut f = Fixture::new().await; + let mut cells = Vec::new(); + for byte in 4..12 { + cells.push(f.cell(byte).await); + } + let mut first_frames = Vec::new(); + let mut first_assignments = Vec::new(); + for cell in &mut cells { + let (_, frames, assigned) = f.append(cell, 2); + first_frames.extend(frames); + first_assignments.push(assigned); + } + let proposal = f + .directory + .prepare_node_bundle(&f.node, &first_frames, &first_assignments, NOW) + .await + .unwrap(); + let (node, proofs) = f + .directory + .select_node_bundle(&f.node, &proposal, &f.lease, Limits::default(), NOW) + .await + .unwrap(); + f.node = node; + for cell in &cells { + let proof = proofs + .iter() + .find(|p| p.binding() == cell.control.value().bundle_binding.unwrap()) + .unwrap(); + f.publisher(cell).materialize_bundle(proof).await.unwrap(); + } + let (_, hot_frames, hot_assigned) = f.append(&mut cells[0], 3); + let proposal = f + .directory + .prepare_node_bundle(&f.node, &hot_frames, &[hot_assigned], NOW) + .await + .unwrap(); + let (node, _) = f + .directory + .select_node_bundle(&f.node, &proposal, &f.lease, Limits::default(), NOW) + .await + .unwrap(); + f.node = node; + let mut frames = Vec::new(); + let mut assignments = Vec::new(); + for (index, cell) in cells.iter_mut().enumerate() { + let (_, capture, assigned) = f.append(cell, if index == 0 { 4 } else { 3 }); + frames.extend(capture); + assignments.push(assigned); + } + let checkpoints = cells + .iter() + .map(|cell| { + let proof = proofs + .iter() + .find(|p| p.binding() == cell.control.value().bundle_binding.unwrap()) + .unwrap(); + (&cell.authority, proof) + }) + .collect::>(); + f.count.reset(); + let proposal = f + .directory + .prepare_node_bundle_with_checkpoints( + &f.node, + &frames, + &assignments, + &checkpoints, + Limits::default(), + NOW, + ) + .await + .unwrap(); + let (node, selected) = f + .directory + .select_node_bundle(&f.node, &proposal, &f.lease, Limits::default(), NOW) + .await + .unwrap(); + assert_eq!(f.count.put_requests(), 2); + f.node = node; + for (index, cell) in cells.iter().enumerate() { + let current = cell + .authority + .load(cell.control.value().cell) + .await + .unwrap() + .unwrap(); + let cold = f + .directory + .load_bundle_coverage(&cell.authority, ¤t, Limits::default()) + .await + .unwrap(); + let live = selected + .iter() + .find(|p| p.binding() == cold.binding()) + .unwrap(); + assert_eq!(cold.commit_sequence(), if index == 0 { 4 } else { 3 }); + assert_eq!(cold.locator_count(), if index == 0 { 2 } else { 1 }); + assert_eq!(live.commit_sequence(), cold.commit_sequence()); + let overlay = cold + .recovery_overlay(&f.layout, Limits::default()) + .await + .unwrap(); + let recovered = cell + .replica + .prepare_recovered_overlay(&overlay, 1) + .await + .unwrap(); + let path = f.scratch.path().join(format!("composed-{index}.sqlite")); + cell.replica + .open_root(&recovered.root()) + .await + .unwrap() + .restore(&path) + .await + .unwrap(); + let db = rusqlite::Connection::open(path).unwrap(); + let count: i64 = db + .query_row("SELECT COUNT(*) FROM outcomes", [], |row| row.get(0)) + .unwrap(); + assert_eq!(count, if index == 0 { 4 } else { 3 }); + let result: String = db + .query_row( + "SELECT result FROM outcomes WHERE request='request-2'", + [], + |row| row.get(0), + ) + .unwrap(); + assert_eq!(result, "result-2"); + } +} + +#[tokio::test] +async fn stale_checkpoint_and_combined_overflow_upload_nothing() { + let mut f = Fixture::new().await; + let mut cell = f.cell(4).await; + let mut proofs = Vec::new(); + for commit in 2..=3 { + let (_, frames, assigned) = f.append(&mut cell, commit); + let proposal = f + .directory + .prepare_node_bundle(&f.node, &frames, &[assigned], NOW) + .await + .unwrap(); + let (node, mut selected) = f + .directory + .select_node_bundle(&f.node, &proposal, &f.lease, Limits::default(), NOW) + .await + .unwrap(); + f.node = node; + proofs.push(selected.pop().unwrap()); + } + f.publisher(&cell) + .materialize_bundle(&proofs[1]) + .await + .unwrap(); + let (_, frames, assigned) = f.append(&mut cell, 4); + f.count.reset(); + assert!(matches!( + f.directory + .prepare_node_bundle_with_checkpoints( + &f.node, + &frames, + &[assigned], + &[(&cell.authority, &proofs[0])], + Limits::default(), + NOW + ) + .await, + Err(Error::PendingPublication) + )); + assert_eq!(f.count.put_requests(), 0); + let checkpoints = vec![(&cell.authority, &proofs[1]); MAX_FRAMES]; + assert!(matches!( + f.directory + .prepare_node_bundle_with_checkpoints( + &f.node, + &frames, + &[assigned], + &checkpoints, + Limits::default(), + NOW + ) + .await, + Err(Error::Capacity(_)) + )); + assert_eq!(f.count.put_requests(), 0); +} + +#[tokio::test] +async fn combined_upload_cannot_select_after_origin_loss_or_original_lease_fencing() { + let mut f = Fixture::new().await; + let mut cell = f.cell(4).await; + let (_, frames, assigned) = f.append(&mut cell, 2); + let proposal = f + .directory + .prepare_node_bundle(&f.node, &frames, &[assigned], NOW) + .await + .unwrap(); + let (node, mut proofs) = f + .directory + .select_node_bundle(&f.node, &proposal, &f.lease, Limits::default(), NOW) + .await + .unwrap(); + f.node = node; + let proof = proofs.pop().unwrap(); + f.publisher(&cell).materialize_bundle(&proof).await.unwrap(); + let (_, frames, assigned) = f.append(&mut cell, 3); + let proposal = f + .directory + .prepare_node_bundle_with_checkpoints( + &f.node, + &frames, + &[assigned], + &[(&cell.authority, &proof)], + Limits::default(), + NOW, + ) + .await + .unwrap(); + let path = f.layout.node_coverage_bundle_path( + f.node.advertisement.session.as_bytes(), + proposal.head.epoch, + proposal.head.digest.as_bytes(), + ); + f.count.reset(); + f.count.block_body_reads_for(&path); + assert!( + f.directory + .select_node_bundle(&f.node, &proposal, &f.lease, Limits::default(), NOW) + .await + .is_err() + ); + assert_eq!(f.count.put_requests(), 0); + f.lease.fence(); + assert!(matches!( + f.directory + .select_node_bundle(&f.node, &proposal, &f.lease, Limits::default(), NOW) + .await, + Err(Error::Fenced) + )); + assert_eq!(f.count.put_requests(), 0); + assert_eq!( + f.directory + .load(f.node.advertisement.session, NOW) + .await + .unwrap() + .unwrap() + .advertisement + .bundle, + f.node.advertisement.bundle + ); +} diff --git a/crates/cellule-runtime/src/node/bundle/tests/index/mod.rs b/crates/cellule-runtime/src/node/bundle/tests/index/mod.rs index 158abc16..3c873e02 100644 --- a/crates/cellule-runtime/src/node/bundle/tests/index/mod.rs +++ b/crates/cellule-runtime/src/node/bundle/tests/index/mod.rs @@ -43,6 +43,7 @@ mod bootstrap; mod checkpoint; mod cohort; mod compatibility; +mod composed; mod copy_on_write; mod density; mod encoding; diff --git a/crates/cellule-runtime/src/node/bundle/tests/managed.rs b/crates/cellule-runtime/src/node/bundle/tests/managed.rs index e4d91a93..6ed81f0c 100644 --- a/crates/cellule-runtime/src/node/bundle/tests/managed.rs +++ b/crates/cellule-runtime/src/node/bundle/tests/managed.rs @@ -91,6 +91,7 @@ impl NodeBundlePublicationAuthority for FailedSelection { fn select<'a>( &'a self, _: &'a [AssignedCapture], + _: &'a [BundleCheckpoint], _: &'a NodeLeaseGuard, ) -> BoxFuture<'a, Result>> { Box::pin(async { Err(Error::Node("injected bundle selection failure")) }) diff --git a/crates/cellule-runtime/src/node/bundle/tests/mod.rs b/crates/cellule-runtime/src/node/bundle/tests/mod.rs index ecd872ce..e0d323c4 100644 --- a/crates/cellule-runtime/src/node/bundle/tests/mod.rs +++ b/crates/cellule-runtime/src/node/bundle/tests/mod.rs @@ -16,6 +16,7 @@ use std::sync::Arc; const NOW: i64 = 1_000_000; const EPOCH: u64 = 2; mod actor; +mod composed; mod coverage; mod faults; mod index; diff --git a/crates/cellule-runtime/src/node/bundle/tests/readiness.rs b/crates/cellule-runtime/src/node/bundle/tests/readiness.rs index 292f6d15..4f009d6d 100644 --- a/crates/cellule-runtime/src/node/bundle/tests/readiness.rs +++ b/crates/cellule-runtime/src/node/bundle/tests/readiness.rs @@ -44,6 +44,7 @@ impl NodeBundlePublicationAuthority for DelayedSelection { fn select<'a>( &'a self, captures: &'a [AssignedCapture], + checkpoints: &'a [BundleCheckpoint], lease: &'a NodeLeaseGuard, ) -> BoxFuture<'a, Result>> { Box::pin(async move { @@ -56,7 +57,7 @@ impl NodeBundlePublicationAuthority for DelayedSelection { changed.await; } } - self.authority.select(captures, lease).await + self.authority.select(captures, checkpoints, lease).await }) } fn checkpoint<'a>(&'a self, checkpoints: &'a [BundleCheckpoint]) -> BoxFuture<'a, Result<()>> { diff --git a/crates/cellule-runtime/src/node/bundle/tests/receipt_pressure.rs b/crates/cellule-runtime/src/node/bundle/tests/receipt_pressure.rs index bfac8ace..ddce6cc2 100644 --- a/crates/cellule-runtime/src/node/bundle/tests/receipt_pressure.rs +++ b/crates/cellule-runtime/src/node/bundle/tests/receipt_pressure.rs @@ -43,10 +43,11 @@ impl NodeBundlePublicationAuthority for HeldReceiptCredit { fn select<'a>( &'a self, captures: &'a [AssignedCapture], + checkpoints: &'a [BundleCheckpoint], lease: &'a NodeLeaseGuard, ) -> BoxFuture<'a, Result>> { Box::pin(async move { - let proofs = self.original.select(captures, lease).await?; + let proofs = self.original.select(captures, checkpoints, lease).await?; if self.selects.fetch_add(1, Ordering::SeqCst) == 1 { let budget = self.ledger.snapshot()?; let remaining = budget.limit.retained_bytes() - budget.used.retained_bytes(); @@ -72,7 +73,7 @@ impl NodeBundlePublicationAuthority for HeldReceiptCredit { } } -fn submission(cell: &mut Cell, commit: u64) -> NodeLogSubmission { +pub(super) fn submission(cell: &mut Cell, commit: u64) -> NodeLogSubmission { cell.db .transaction(|tx| { tx.execute( diff --git a/crates/cellule-runtime/src/node/durability/publication/checkpoints.rs b/crates/cellule-runtime/src/node/durability/publication/checkpoints.rs new file mode 100644 index 00000000..226cde9c --- /dev/null +++ b/crates/cellule-runtime/src/node/durability/publication/checkpoints.rs @@ -0,0 +1,63 @@ +//! Bounded original callback ownership shared by combined and idle publication. +use super::*; + +pub(super) struct Cohort { + values: Vec, + completions: Vec>, +} + +impl Cohort { + pub(super) fn gather( + receiver: &mut mpsc::Receiver, + first: Option, + maximum: usize, + ) -> Result { + let mut cohort = Self { + values: Vec::new(), + completions: Vec::new(), + }; + if let Some(first) = first { + if maximum == 0 { + return Err(Error::Capacity("combined checkpoint count")); + } + cohort.push(first)?; + } + while cohort.completions.len() < maximum { + let Ok(next) = receiver.try_recv() else { + break; + }; + cohort.push(next)?; + } + Ok(cohort) + } + + fn push(&mut self, request: CheckpointRequest) -> Result<()> { + self.completions.push(request.completed); + let next = request.checkpoint; + if let Some(index) = self + .values + .iter() + .position(|c| c.proof().binding() == next.proof().binding()) + { + if self.values[index].root.commit_sequence >= next.root.commit_sequence { + return Err(Error::Node("bundle checkpoint order regressed")); + } + // Keep every original callback even when its newer proof supersedes + // the row. All callbacks join the same verified catalog selection. + self.values[index] = next; + } else { + self.values.push(next); + } + Ok(()) + } + + pub(super) fn values(&self) -> &[BundleCheckpoint] { + &self.values + } + + pub(super) fn complete(self, result: PublicationResult) { + for completion in self.completions { + let _ = completion.send(result.clone()); + } + } +} diff --git a/crates/cellule-runtime/src/node/durability/publication/mod.rs b/crates/cellule-runtime/src/node/durability/publication/mod.rs index eed86100..7d94e6a2 100644 --- a/crates/cellule-runtime/src/node/durability/publication/mod.rs +++ b/crates/cellule-runtime/src/node/durability/publication/mod.rs @@ -5,6 +5,8 @@ use crate::node::log_shipper::{AssignedCapture, NodePublicationFeed, SelectedBun use std::sync::{Mutex as StdMutex, Weak}; use tokio::sync::{mpsc, oneshot, watch}; +mod checkpoints; + const MAX_CAPTURES: usize = 64; const MAX_FRAMES: usize = 64; const MAX_NATIVE_BYTES: usize = crate::node::bundle::MAX_BUNDLE_BYTES as usize; @@ -38,10 +40,13 @@ impl BundleCheckpoint { /// Implement with the same original state/heartbeat mutex as binding and close. pub trait NodeBundlePublicationAuthority: NodeBundleAuthority { /// Selects the complete ordered cohort with its native coverage in one CAS. - /// Return original live proofs only after dependency verification and CAS. + /// Includes ready exact materialized checkpoints in that same catalog/CAS. + /// Return original live proofs only after dependency verification and CAS; + /// success also joins every supplied checkpoint obligation. fn select<'a>( &'a self, captures: &'a [AssignedCapture], + checkpoints: &'a [BundleCheckpoint], lease: &'a NodeLeaseGuard, ) -> BoxFuture<'a, Result>>; /// Checkpoints exact materialized roots together. An obsolete notification @@ -107,7 +112,10 @@ impl Publisher { let _working = working; let result = run(weak, authority, feed, receiver, &progress, &running_lease) .await - .map_err(Arc::new); + .map_err(|error| match error { + Error::Shared(source) => source, + error => Arc::new(error), + }); progress.send_modify(|state| state.terminal = Some(result.clone())); if result.is_err() { running_lease.fence(); @@ -218,19 +226,19 @@ async fn run( let mut checkpoints_open = true; loop { lease.check()?; - // One checkpoint cohort gets a turn even when native carryover never - // empties. Then prefer already queued native work over another root - // cohort; idle producers can still join all outstanding checkpoints. - if let Ok(first) = checkpoints.try_recv() { - tokio::select! { - result = checkpoint_cohort(&authority, &mut checkpoints, first) => result?, - () = lease.wait_fenced() => return Err(Error::Fenced), - } - } + // Retain one ready root notification so native work can share its + // catalog/CAS. Idle publication still joins checkpoints immediately. + let mut first_checkpoint = checkpoints.try_recv().ok(); let capture = if carry.is_some() { carry.take() } else if let Some(capture) = feed.try_recv() { Some(capture) + } else if let Some(first) = first_checkpoint.take() { + tokio::select! { + result = checkpoint_cohort(&authority, &mut checkpoints, first) => result?, + () = lease.wait_fenced() => return Err(Error::Fenced), + } + continue; } else { tokio::select! { capture = feed.recv() => capture, @@ -248,6 +256,9 @@ async fn run( } }; let Some(first) = capture else { + if let Some(first) = first_checkpoint.take() { + checkpoint_cohort(&authority, &mut checkpoints, first).await?; + } break; }; let mut captures = vec![first]; @@ -258,6 +269,15 @@ async fn run( "complete capture exceeds native bundle bounds", )); } + if frames == MAX_FRAMES { + if let Some(first) = first_checkpoint.take() { + tokio::select! { + result = checkpoint_cohort(&authority, &mut checkpoints, first) => result?, + () = lease.wait_fenced() => return Err(Error::Fenced), + } + } + } + let reserved_checkpoint = usize::from(first_checkpoint.is_some()); let deadline = tokio::time::Instant::now() + ASSEMBLY; while captures.len() < MAX_CAPTURES { let next = tokio::select! { @@ -269,7 +289,9 @@ async fn run( break; }; let next_bytes = capture_bytes(&next)?; - if frames + next.frames().len() > MAX_FRAMES || bytes + next_bytes > MAX_NATIVE_BYTES { + if frames + next.frames().len() + reserved_checkpoint > MAX_FRAMES + || bytes + next_bytes > MAX_NATIVE_BYTES + { carry = Some(next); break; } @@ -277,17 +299,29 @@ async fn run( bytes += next_bytes; captures.push(next); } + let ready_checkpoints = + checkpoints::Cohort::gather(&mut checkpoints, first_checkpoint, MAX_FRAMES - frames)?; let original = durability.upgrade().ok_or(Error::RuntimeClosed)?; - let proofs = tokio::select! { - proofs = authority.select(&captures, lease) => proofs?, - () = lease.wait_fenced() => return Err(Error::Fenced), - }; + let result = tokio::select! { + proofs = authority.select(&captures, ready_checkpoints.values(), lease) => proofs, + () = lease.wait_fenced() => Err(Error::Fenced), + } + .map_err(Arc::new); + // Root tasks can release their original credit before receipt admission. + // Completion follows the combined canonical CAS, never enqueue/upload. + ready_checkpoints.complete(result.as_ref().map(|_| ()).map_err(Arc::clone)); + let proofs = result.map_err(Error::Shared)?; let cohort = receipts::SelectedCaptures::new(&original, &captures, proofs)?; let memory = { let admission = cohort.resources(&original)?.reserve(cohort.cost()); tokio::pin!(admission); loop { tokio::select! { + // A ready original credit must win over an unrelated root + // callback. Only actual pressure requires standalone work; + // otherwise that callback can share the next native CAS. + biased; + () = lease.wait_fenced() => return Err(Error::Fenced), result = &mut admission => break result?, checkpoint = checkpoints.recv(), if checkpoints_open => { if let Some(first) = checkpoint { @@ -300,7 +334,6 @@ async fn run( } } else { checkpoints_open = false; } } - () = lease.wait_fenced() => return Err(Error::Fenced), } } }; @@ -332,29 +365,11 @@ async fn checkpoint_cohort( receiver: &mut mpsc::Receiver, first: CheckpointRequest, ) -> Result<()> { - let mut completions = vec![first.completed]; - let mut cohort = vec![first.checkpoint]; - while completions.len() < MAX_CAPTURES { - let Ok(next) = receiver.try_recv() else { - break; - }; - completions.push(next.completed); - let next = next.checkpoint; - if let Some(index) = cohort - .iter() - .position(|c| c.proof().binding() == next.proof().binding()) - { - if cohort[index].root.commit_sequence >= next.root.commit_sequence { - return Err(Error::Node("bundle checkpoint order regressed")); - } - cohort[index] = next; - } else { - cohort.push(next); - } - } - let result = authority.checkpoint(&cohort).await.map_err(Arc::new); - for completion in completions { - let _ = completion.send(result.clone()); - } + let cohort = checkpoints::Cohort::gather(receiver, Some(first), MAX_CAPTURES)?; + let result = authority + .checkpoint(cohort.values()) + .await + .map_err(Arc::new); + cohort.complete(result.clone()); result.map_err(Error::Shared) } From beda5dbf99c540e92d017c5b434fae2218d5a8d0 Mon Sep 17 00:00:00 2001 From: forhappy Date: Fri, 9 Oct 2026 17:54:21 -0700 Subject: [PATCH 100/102] Use a compound guard for full native captures --- .../src/node/durability/publication/mod.rs | 12 ++++++------ 1 file changed, 6 insertions(+), 6 deletions(-) diff --git a/crates/cellule-runtime/src/node/durability/publication/mod.rs b/crates/cellule-runtime/src/node/durability/publication/mod.rs index 7d94e6a2..cab6b7c5 100644 --- a/crates/cellule-runtime/src/node/durability/publication/mod.rs +++ b/crates/cellule-runtime/src/node/durability/publication/mod.rs @@ -269,12 +269,12 @@ async fn run( "complete capture exceeds native bundle bounds", )); } - if frames == MAX_FRAMES { - if let Some(first) = first_checkpoint.take() { - tokio::select! { - result = checkpoint_cohort(&authority, &mut checkpoints, first) => result?, - () = lease.wait_fenced() => return Err(Error::Fenced), - } + if frames == MAX_FRAMES + && let Some(first) = first_checkpoint.take() + { + tokio::select! { + result = checkpoint_cohort(&authority, &mut checkpoints, first) => result?, + () = lease.wait_fenced() => return Err(Error::Fenced), } } let reserved_checkpoint = usize::from(first_checkpoint.is_some()); From 80caaeab6d935a93b16803475a98449beb52e265 Mon Sep 17 00:00:00 2001 From: forhappy Date: Fri, 9 Oct 2026 18:38:01 -0700 Subject: [PATCH 101/102] Reserve ready checkpoints before filling native publication batches --- .../src/node/bundle/tests/composed.rs | 180 +++++++++++++++++- .../durability/publication/checkpoints.rs | 19 +- .../src/node/durability/publication/mod.rs | 12 +- 3 files changed, 203 insertions(+), 8 deletions(-) diff --git a/crates/cellule-runtime/src/node/bundle/tests/composed.rs b/crates/cellule-runtime/src/node/bundle/tests/composed.rs index cbe13c0e..99bdb977 100644 --- a/crates/cellule-runtime/src/node/bundle/tests/composed.rs +++ b/crates/cellule-runtime/src/node/bundle/tests/composed.rs @@ -14,6 +14,8 @@ struct HeldSelection { original: Arc, calls: AtomicUsize, combined: AtomicUsize, + largest_combined: AtomicUsize, + hold_after: usize, entered: tokio::sync::Notify, resume: tokio::sync::Notify, } @@ -44,7 +46,9 @@ impl NodeBundlePublicationAuthority for HeldSelection { Box::pin(async move { let result = self.original.select(captures, checkpoints, lease).await?; self.combined.fetch_add(checkpoints.len(), Ordering::SeqCst); - if self.calls.fetch_add(1, Ordering::SeqCst) == 1 { + self.largest_combined + .fetch_max(checkpoints.len(), Ordering::SeqCst); + if self.calls.fetch_add(1, Ordering::SeqCst) == self.hold_after { self.entered.notify_one(); self.resume.notified().await; } @@ -83,6 +87,8 @@ async fn producer_combines_ready_callback_with_queued_native_work_and_joins_comp original, calls: AtomicUsize::new(0), combined: AtomicUsize::new(0), + largest_combined: AtomicUsize::new(0), + hold_after: 1, entered: tokio::sync::Notify::new(), resume: tokio::sync::Notify::new(), }); @@ -177,3 +183,175 @@ async fn producer_combines_ready_callback_with_queued_native_work_and_joins_comp ); pool.shutdown().await.unwrap(); } + +#[tokio::test] +async fn producer_reserves_the_ready_checkpoint_cohort_before_filling_native_capacity() { + let mut f = Fixture::new().await; + super::coverage::enroll(&mut f).await; + let mut cells = Vec::new(); + for byte in 4..12 { + cells.push(f.cell(byte).await); + } + let original = Arc::new(Authority { + directory: f.directory.clone(), + observed: tokio::sync::Mutex::new(f.node.clone()), + }); + let peers = transport(&f, true); + let shipper = NodeLogShipper::new(f.gate.clone(), peers.clone(), Limits::default()).unwrap(); + let durability = Arc::new(NodeDurability::new( + f.gate.clone(), + shipper, + original.clone(), + peers, + f.lease.clone(), + )); + let pool = SqlWorkerPool::new(1, 1).unwrap(); + pool.configure_retained_capacity(64 << 20).unwrap(); + durability + .attach_selection_resources(pool.resource_ledger()) + .unwrap(); + let held = Arc::new(HeldSelection { + original, + calls: AtomicUsize::new(0), + combined: AtomicUsize::new(0), + largest_combined: AtomicUsize::new(0), + hold_after: cells.len(), + entered: tokio::sync::Notify::new(), + resume: tokio::sync::Notify::new(), + }); + durability.start_bundle_publication(held.clone()).unwrap(); + let mut roots = Vec::new(); + for cell in &mut cells { + let pending = durability + .submit_capture(submission(cell, 2)) + .await + .unwrap(); + let selected = pending + .selection + .as_ref() + .unwrap() + .selected() + .await + .unwrap(); + let root = f + .publisher(cell) + .materialize_bundle(&selected.proof) + .await + .unwrap(); + roots.push((root, selected)); + } + let pending = durability + .submit_capture(submission(&mut cells[0], 3)) + .await + .unwrap(); + tokio::time::timeout(std::time::Duration::from_secs(5), held.entered.notified()) + .await + .unwrap(); + let mut commits = vec![2; cells.len()]; + commits[0] = 3; + let mut queued = Vec::new(); + // More native work than one bounded selection can consume. Every ready + // original root callback must still fit in the next combined selection. + for index in 0..64 { + let cell_index = index % cells.len(); + commits[cell_index] += 1; + queued.push( + durability + .submit_capture(submission(&mut cells[cell_index], commits[cell_index])) + .await + .unwrap(), + ); + } + let mut callbacks = cells + .iter() + .zip(&roots) + .map(|(cell, (root, selected))| { + Box::pin(tokio::task::unconstrained( + durability.checkpoint_materialized(cell.authority.clone(), *root, selected.clone()), + )) + }) + .collect::>(); + for callback in &mut callbacks { + assert!(poll!(callback.as_mut()).is_pending()); + } + held.resume.notify_one(); + tokio::time::timeout(std::time::Duration::from_secs(5), async { + for callback in &mut callbacks { + callback.await.unwrap(); + } + pending + .selection + .as_ref() + .unwrap() + .selected() + .await + .unwrap(); + for capture in &queued { + capture + .selection + .as_ref() + .unwrap() + .selected() + .await + .unwrap(); + } + }) + .await + .unwrap(); + assert_eq!(held.largest_combined.load(Ordering::SeqCst), cells.len()); + let mut last = Vec::new(); + for cell_index in 0..cells.len() { + let selected = queued[56 + cell_index] + .selection + .as_ref() + .unwrap() + .selected() + .await + .unwrap(); + assert_eq!(selected.proof.base().unwrap(), roots[cell_index].0); + assert_eq!(selected.proof.commit_sequence(), commits[cell_index]); + last.push(selected); + } + for (cell, selected) in cells.iter_mut().zip(&last) { + cell.control = cell + .authority + .load(cell.control.value().cell) + .await + .unwrap() + .unwrap(); + let root = f + .publisher(cell) + .materialize_bundle(&selected.proof) + .await + .unwrap(); + durability + .checkpoint_materialized(cell.authority.clone(), root, selected.clone()) + .await + .unwrap(); + cell.control = cell + .authority + .load(cell.control.value().cell) + .await + .unwrap() + .unwrap(); + durability + .close_bundle_cell(&cell.authority, &cell.control) + .await + .unwrap(); + } + durability.shutdown().await.unwrap(); + drop(pending); + drop(queued); + drop(callbacks); + drop(roots); + drop(last); + assert_eq!( + pool.resource_ledger() + .snapshot() + .unwrap() + .used + .retained_bytes(), + 0 + ); + pool.shutdown().await.unwrap(); +} diff --git a/crates/cellule-runtime/src/node/durability/publication/checkpoints.rs b/crates/cellule-runtime/src/node/durability/publication/checkpoints.rs index 226cde9c..5138b216 100644 --- a/crates/cellule-runtime/src/node/durability/publication/checkpoints.rs +++ b/crates/cellule-runtime/src/node/durability/publication/checkpoints.rs @@ -22,13 +22,26 @@ impl Cohort { } cohort.push(first)?; } - while cohort.completions.len() < maximum { + cohort.extend(receiver, maximum)?; + Ok(cohort) + } + + pub(super) fn extend( + &mut self, + receiver: &mut mpsc::Receiver, + maximum: usize, + ) -> Result<()> { + while self.completions.len() < maximum { let Ok(next) = receiver.try_recv() else { break; }; - cohort.push(next)?; + self.push(next)?; } - Ok(cohort) + Ok(()) + } + + pub(super) fn notification_count(&self) -> usize { + self.completions.len() } fn push(&mut self, request: CheckpointRequest) -> Result<()> { diff --git a/crates/cellule-runtime/src/node/durability/publication/mod.rs b/crates/cellule-runtime/src/node/durability/publication/mod.rs index cab6b7c5..24fcb4be 100644 --- a/crates/cellule-runtime/src/node/durability/publication/mod.rs +++ b/crates/cellule-runtime/src/node/durability/publication/mod.rs @@ -277,7 +277,12 @@ async fn run( () = lease.wait_fenced() => return Err(Error::Fenced), } } - let reserved_checkpoint = usize::from(first_checkpoint.is_some()); + // Original root tasks hold credit until their callback joins. Reserve + // the entire ready cohort before native assembly can occupy its rows; + // a continuously full native feed must not split it into tiny CASes. + let mut ready_checkpoints = + checkpoints::Cohort::gather(&mut checkpoints, first_checkpoint, MAX_FRAMES - frames)?; + let reserved_checkpoints = ready_checkpoints.notification_count(); let deadline = tokio::time::Instant::now() + ASSEMBLY; while captures.len() < MAX_CAPTURES { let next = tokio::select! { @@ -289,7 +294,7 @@ async fn run( break; }; let next_bytes = capture_bytes(&next)?; - if frames + next.frames().len() + reserved_checkpoint > MAX_FRAMES + if frames + next.frames().len() + reserved_checkpoints > MAX_FRAMES || bytes + next_bytes > MAX_NATIVE_BYTES { carry = Some(next); @@ -299,8 +304,7 @@ async fn run( bytes += next_bytes; captures.push(next); } - let ready_checkpoints = - checkpoints::Cohort::gather(&mut checkpoints, first_checkpoint, MAX_FRAMES - frames)?; + ready_checkpoints.extend(&mut checkpoints, MAX_FRAMES - frames)?; let original = durability.upgrade().ok_or(Error::RuntimeClosed)?; let result = tokio::select! { proofs = authority.select(&captures, ready_checkpoints.values(), lease) => proofs, From 6a93cbab73f6316e1fb4c4c33d18f94901361e49 Mon Sep 17 00:00:00 2001 From: forhappy Date: Fri, 9 Oct 2026 19:15:25 -0700 Subject: [PATCH 102/102] Record composed publication measurements and qualification gaps --- .../docs/write-performance-design.md | 19 ++- docs/pr67-composed-publication-measurement.md | 159 ++++++++++++++++++ docs/write-performance-delivery.md | 10 ++ 3 files changed, 182 insertions(+), 6 deletions(-) create mode 100644 docs/pr67-composed-publication-measurement.md diff --git a/crates/cellule-runtime/docs/write-performance-design.md b/crates/cellule-runtime/docs/write-performance-design.md index c5c27e13..5f90ac8c 100644 --- a/crates/cellule-runtime/docs/write-performance-design.md +++ b/crates/cellule-runtime/docs/write-performance-design.md @@ -1,11 +1,18 @@ # Node write and read performance design -The current candidate composes ready materialized checkpoints and new native -captures into one fresh catalog load, upload and fenced selection CAS. Exact -prefix proofs, original callbacks, the 64-row limit and the 20-MiB publication -working reservation remain in force. Idle and receipt-pressure paths still join -standalone checkpoints. Measurement and qualification determine whether this -candidate is retained; the diagnostic history below remains separate evidence. +The latest [composed-publication measurement](../../../docs/pr67-composed-publication-measurement.md) +joins ready exact materialized checkpoints and new complete native captures in +one fresh bounded catalog load, upload and fenced CAS. Reserving all ready +checkpoint rows before native assembly fixes a demonstrated cohort split. Fresh +Fleet TPS rises 407.03→468.80/s; successful scheduled p99 stays about 389 ms. All +53,527 candidate ACKs pass warm/cold mutation and original retry audits, and all +2,000 Cells drain idle. Node-authority PUTs/success fall 21.38%, but total observed +store bytes/success rise 8.68% and publication debt still grows. Errors/drops, +latency, sustained capacity, streaming transport and complete qualification remain +open. This short observation establishes no repeatable performance parity; PR +#67 stays a draft and all acceptance gates remain unchanged. + +The following measurements remain historical observations. The latest [shared-overlay diagnostic](../../../docs/pr67-shared-overlay-measurement.md) rejects and reverts recovered-overlay integration through the shared producer. diff --git a/docs/pr67-composed-publication-measurement.md b/docs/pr67-composed-publication-measurement.md new file mode 100644 index 00000000..dbe07f8d --- /dev/null +++ b/docs/pr67-composed-publication-measurement.md @@ -0,0 +1,159 @@ +# PR 67: compose checkpoints with native publication + +**Write parity remains unmet; PR #67 remains a draft.** The corrected candidate +completes 468.80 successful Fleet writes/s versus 407.03/s in its fresh baseline. +Successful scheduled p99 is unchanged at about 389 ms. The candidate passes all +53,527 warm/cold mutation and original retry checks and joins its drain. Errors, +dropped offers, growing publication debt and expensive historical reads still +fail performance acceptance. This short pair establishes an observation, not a +repeatable capacity improvement. + +## Implemented behavior + +Ready exact materialized checkpoints and new complete captures now share one +fresh catalog load, upload and fenced native-selection CAS. The canonical +checkpoint validator checks original authority, scope, pin, root endpoint and +identical historical prefix before pruning only the covered locators. Native +selection retains fresh origin matching, base/history verification and original +lease checks. No origin-availability cache crosses operations or grants an ACK. + +The producer reserves rows for the entire ready callback cohort before filling +the native batch. All original callbacks complete after successful canonical CAS, +before receipt admission; coalescing newer same-pin notifications retains every +callback. The bound remains 64 native frames plus checkpoint notifications, with +4 MiB native bytes and the existing 20 MiB working reservation. Idle, full-capture, +actual receipt-pressure and shutdown paths still join standalone checkpoints. +Ready receipt credit wins over unrelated callbacks; lease fencing wins over both. + +This extends the existing admission and shipping changes in the +[native pipeline](pr67-native-pipeline-measurement.md): publication capacity waits +precede global ordering, eight original rounds use independent ordered member +lanes, and delivered frames use canonical follower group commit. Each member +still awaits one grouped RPC at a time. Celld's +[ordered shipping](https://github.com/denoland/celld/blob/f2bf648663a610eefde71f3547ad61e9b896b1f0/crates/celld/ltx_repl.rs#L4530), +[ordered receiver/group commit](https://github.com/denoland/celld/blob/f2bf648663a610eefde71f3547ad61e9b896b1f0/crates/celld/node_log.rs#L179) +and [new-entry upload](https://github.com/denoland/celld/blob/f2bf648663a610eefde71f3547ad61e9b896b1f0/crates/celld/node_log.rs#L6628) +remain architectural references; Cellule does not yet implement their full cost +and concurrency model. No celld source is copied in this change. + +## Fresh matched diagnostics + +Each case uses 2,000 uniform Cells, 96-byte SQL values, the same persisted +request/result ledger, 128 clients and queue slots, one owner and two followers, +mTLS, WAL NORMAL, tmpfs and pinned RustFS. Offered load is 2,000 writes/s with no +reads, 30-second warmup and a 60-second measured window. Driver, auditor, fixture +bytes, images and workload manifests match. All original build and test handles +join before timed runs; independent replay runs after the complete run controllers +join. No contributor suite, build or independent replay overlaps timed windows. + +Baseline serving revision `cf4785c675698afe786f566242d6d3dace32e775` has production +bytes identical to retained `a4ad3401e4d398363c54d000a279d010ddba3abc`. Initial +composed candidate `beda5dbf99c540e92d017c5b434fae2218d5a8d0` reserves only one ready +checkpoint row before native assembly. Corrected candidate +`80caaeab6d935a93b16803475a98449beb52e265` reserves the entire ready cohort. Each +comparison has a freshly executed baseline and celld `f2bf6486`; the corrected +run order is candidate, baseline, celld. Neither comparison is a repeated A/A +variance study. Documentation-only delivery commits preserve measured source. + +| Comparison | Arm | Successful writes/s | Successful scheduled p99 ms | Errors | Dropped offers | +| --- | --- | ---: | ---: | ---: | ---: | +| Initial | Cellule baseline | 364.02 | 261.94 | 79,807 | 18,251 | +| Initial | Composed candidate | 328.98 | 334.94 | 61,020 | 39,239 | +| Initial | celld | 1,949.30 | 213.77 | 0 | 3,012 | +| Corrected | Fresh Cellule baseline | 407.03 | 389.17 | 53,131 | 42,447 | +| Corrected | Ready-cohort candidate | 468.80 | 389.62 | 42,229 | 49,517 | +| Corrected | Fresh celld | 1,973.50 | 34.00 | 0 | 1,585 | + +TPS counts successful completions inside the measured window. Successful +scheduled p99 is reconstructed from successful measured offers through client +drain, excluding fast failures; request p99 for the corrected baseline/candidate +is 209.50/209.38 ms. The initial candidate loses 9.62% TPS and worsens p99 by +27.87%. The corrected candidate gains 15.17% observed TPS over its fresh baseline; +p99 rises 0.12%, so there is no demonstrated latency improvement. The same +baseline binary yields 364.02 and 407.03/s, and celld p99 varies substantially: +these runs cannot establish a repeatable causal gain or maximum sustainable TPS. + +Independent replay reconciles all planned offers, attempts, errors, drops, +trailing completions, per-Cell successes, payload bytes, complete ACK provenance +and every original successful response. All 2,000 Cells have measured successes. +The corrected candidate has 28,128 in-window successes and 126 trailing successes. + +| Comparison/arm | Complete ACK cohort | Warm audit | Cold audit and joined drain | +| --- | ---: | --- | --- | +| Initial baseline | 42,864 | 12,573 HTTP 503s | Not reached | +| Initial candidate | 41,984 | All mutations/retries pass | All pass; 71.51 s | +| Initial celld | 178,731 | All mutations/retries pass | All pass; 25.49 s | +| Corrected baseline | 49,878 | 187 HTTP 503s | Not reached; owner cleanup times out | +| Corrected candidate | 53,527 | All mutations/retries pass | All pass; 54.80 s | +| Corrected celld | 180,416 | All mutations/retries pass | All pass; 19.85 s | + +HTTP 503 audit failures establish failed availability, not acknowledged-data loss. +Both Cellule candidates complete exact cold restoration and all original retries. +No case passes qualification; even celld drops offers in these runs. + +## Cost and debt findings + +In the corrected comparison, node-authority PUT starts per successful write fall +0.08017 to 0.06303 (21.38%), while total observed store bytes per success rise +72,635 to 78,940 (8.68%). Node authority accounts for 60,269/66,650 bytes per +success. All operation families, including reads and coordination, are included; +these ratios normalize a concurrent window and are not per-request tracing. +Fewer catalog transactions do not establish lower total publication cost. + +Native submission's seven phase totals reconcile with its complete phase cohort. +Mean global-order wait is 0.00106/0.00050 ms for baseline/candidate; publication-slot +wait is 0.00040/0.00036 ms. The separate Fleet-proof cohort averages 69.48/71.34 ms, +and follower frames/sync are about 11.39/10.54. These overlapping caller waits +cannot be added as serial service time or attributed causally to batching. + +The corrected candidate retains 2,000 active Cells and logs no owner-capacity +fences; the baseline falls to 1,991 and logs nine. All candidate Cells drain idle; +the baseline warm audit fails and cleanup does not join normally. Candidate +pending publication count falls 2,973 to 2,102 and retained captures fall 2.45 to +0.71 MB, but oldest age rises 51.74 to 62.80 seconds and unpublished node-log bytes +rise 28.33 to 47.17 MB. Retained RAM rises 53.63 to 64.52 MB. Endpoint observations +and a short window do not prove bounded sustained debt. Shared root-packing +counters are zero: materialized roots remain per Cell. + +## Verification and remaining delivery + +All contributor gates pass on an isolated snapshot of the measured source: all +features/targets, full workspace tests, local LTX, Rust 1.97/1.99 Clippy with +warnings denied, API docs, boundaries/layout, Rust fences/links, SQL/peer +contracts and script tests. The workspace reports 1,999 passing test/doctest +executions (including child reports), zero failures and 38 environment-dependent +ignores; local LTX reports 60 passes. All 97 bundle tests, 18 shipping tests and +three repeated six-test composed suites pass. Final documentation changes +receive separate syntax/link/format checks and preserve measured production bytes. + +The component fixture first fails with four PUTs and passes with two, with exact +expected-root/cold-image equality. Additional tests cover eight independent +roots, retained later suffixes, stale endpoints, count overflow, origin loss and +lease fencing. The real producer test joins queued checkpoints, complete native +ranges and shutdown with zero retained credit. The saturated-native test first +fails with a largest cohort of seven out of eight ready callbacks; the corrected +scheduler joins all eight in one selection and passes repeated runs without +changing its deadline or recovery assertions. The initial experiment and external +analyzer-name-filter failures remain recorded separately; production, workload, +profiles and expected evidence are not weakened. + +This is an 8-vCPU/approximately-8-GiB shared Docker VM, not a dedicated owner with +8 CPUs and 16 GiB plus isolated followers/store. Container ceilings of 8 CPUs and +16 GiB do not supply those resources. Cellule's 64 MiB retained and 1 GiB managed +disk budgets remain unchanged; celld's internal budgets are not matched. Tmpfs +and NORMAL do not qualify physical device/power-loss durability. These limits +prevent extrapolation to the laptop KV result or production capacity. + +Remaining delivery prioritizes lower total catalog/history bytes and verification +work using exact authenticated proofs, measured bounded materialization admission, +ordered transport with multiple frames in flight and follower group commit, and +warm overload availability. Acceptance still requires three matched repetitions +of at least five minutes, zero errors/drops, target latency/throughput, bounded +memory/debt, all-ACK fault recovery and read-only/mixed guardrails. Read throughput +and fault qualification are not measured by these write-only diagnostics. + +Raw evidence, failed attempts, source manifests, binary hashes and original logs +remain outside Git at +`/Volumes/Workspace/crabbuild-target/native-composed-selection-20261009`. +The complete frozen hash index records their provenance; mutable compiler caches +and disposable independent SQLite replay indices are excluded. diff --git a/docs/write-performance-delivery.md b/docs/write-performance-delivery.md index 8ebf1700..23b9cedb 100644 --- a/docs/write-performance-delivery.md +++ b/docs/write-performance-delivery.md @@ -1,5 +1,15 @@ # Write performance implementation and verification +The latest [composed-publication diagnostic](pr67-composed-publication-measurement.md) +reserves the complete ready root-callback cohort before native batching and +shares one exact catalog/CAS with new captures. Its fresh corrected comparison +records 407.03→468.80 successful Fleet writes/s with unchanged ~389-ms successful +scheduled p99; every one of 53,527 ACKs passes warm/cold mutation and retry +audits. The initial composed candidate regresses 364.02→328.98/s and remains +recorded separately. Total store bytes/success and publication debt still grow; +errors/drops and the remaining transport/qualification gaps keep PR #67 a draft. +Earlier milestone reports below are historical; acceptance gates are unchanged. + This delivers packed dependencies, shared publication, signed append grants and a quantified gap report, not the completed M0–M5 plan. A pinned Docker comparison retains exact retry and cold-state audits. **Celld write parity