diff --git a/src/api/router/snapshot_chunks_bounded_tests.rs b/src/api/router/snapshot_chunks_bounded_tests.rs index 8d22a68d..243d326a 100644 --- a/src/api/router/snapshot_chunks_bounded_tests.rs +++ b/src/api/router/snapshot_chunks_bounded_tests.rs @@ -214,6 +214,14 @@ async fn mst2_large_chunk_uses_current_oid_strict_range_faults_cancel_retry_and_ fixture.counts.assert(1, size as usize); let map_id = map["map"]["map_id"].as_str().unwrap(); fixture.counts.reset(); + super::persisted_chunk_maps::assert_three_page_proofs_and_selected_sibling_faults( + &fixture, map_id, digest, &pattern, + ) + .await; + // Each distinct OID earns its own full-stream receipt before map reuse. + assert_eq!(fixture.map("/other").await["map"], map["map"]); + fixture.counts.assert(1, size as usize); + fixture.counts.reset(); // Equal content map sharing never carries the first source's OID. let body = request("/other", digest, map_id, 512); assert_chunk( @@ -387,6 +395,26 @@ async fn mst2_chunk_batch_live_budget_and_invalid_later_path_reject_before_body_ ], ) .await; + let budget = crate::ceres::snapshot::content_budget::MemoryBudget::new(1024 * 1024); + let repository = crate::jupiter::storage::native_chunk_map::PostgresChunkMapRepository::new( + fixture + .state + .storage + .mono_storage() + .get_connection() + .clone(), + ) + .await + .unwrap() + .with_test_budget(budget.clone()); + assert!( + fixture + .state + .storage + .native_chunk_maps + .set(repository) + .is_ok() + ); let mut items = Vec::new(); for (index, path) in ["/file", "/one", "/two"].iter().enumerate() { let oid = oid_for(&fixture, path).await; @@ -417,6 +445,7 @@ async fn mst2_chunk_batch_live_budget_and_invalid_later_path_reject_before_body_ ) .await; fixture.counts.assert(0, 0); + assert_eq!(budget.used(), 0); let fixture = Fixture::new().await; let mut body = fixture.chunk_body("/file", &format!("sha256:{}", hex_of(&[1; 32])), "0"); let mut invalid = body["items"][0].clone(); diff --git a/src/api/router/snapshot_content.rs b/src/api/router/snapshot_content.rs index ea7d8938..d642434f 100644 --- a/src/api/router/snapshot_content.rs +++ b/src/api/router/snapshot_content.rs @@ -4,10 +4,9 @@ //! WP and `frame_encodings` advertises identity alone. use axum::{ - Json, extract::{Path as AxumPath, Query, State}, http::HeaderMap, - response::{IntoResponse, Response}, + response::Response, }; use futures::stream::StreamExt; use serde::Deserialize; @@ -19,8 +18,8 @@ use super::{ mst2_error_response, request::Mst2Bytes, }; use crate::ceres::snapshot::{ - chunks::{ChunkProjection, get_or_project_stream, projection_reservation_bytes}, - content_budget::{PROJECTION_LIVE_BYTES, reserve_range_work, reserve_response}, + chunks::{ChunkMapSource, VerifiedSourceChunkMap, map_build_reservation_bytes}, + content_budget::{BudgetedFrame, MemoryLease, reserve_range_work, reserve_response}, error::{SnapshotError, SnapshotErrorCode}, pages::{MetadataWalkOutcome, base64_of, hex_of, resolve_abs_metadata}, resolver::FsKind, @@ -32,6 +31,7 @@ pub(super) struct ResolvedFileMetadata { oid: String, pub(super) digest: [u8; 32], pub(super) size: u64, + fact: crate::callisto::mst2_verified_object::Model, } #[allow(clippy::result_large_err)] @@ -154,6 +154,7 @@ pub(super) async fn verified_file_metadata, #[serde(default)] - pub(super) page: Option, - /// Canonical v3 page requests bind the page to the already verified map - /// instead of repeating the file digest. - #[serde(default)] pub(super) map_id: Option, #[serde(default)] pub(super) page_index: Option, @@ -475,7 +472,8 @@ async fn project_for( scope: &str, path: &str, expected_digest: Option<&str>, -) -> Result, Response> { +) -> Result, Response> +{ let f = resolve_file_metadata(handler, root_tree, scope, path, expected_digest).await?; project_resolved(handler, &f).await } @@ -484,36 +482,61 @@ async fn project_for( async fn project_resolved( handler: &T, f: &ResolvedFileMetadata, -) -> Result, Response> { - // The first request for a digest builds the projection from the fixed - // Git object; later requests slice the cached representation. A miss - // rebuilds, never errors with "missing chunk". - let digest = f.digest; - let size = f.size; - let projection = get_or_project_stream(digest, size, || async move { - handler - .get_raw_blob_stream_by_hash(&f.oid) - .await - .map_err(content_read_error) - }) - .await - .map_err(|error| { - mst2_error_response(if error.code == SnapshotErrorCode::DigestMismatch { - SnapshotError::new( - SnapshotErrorCode::IntegrityError, - "fixed blob digest disagrees with its verified fact", - ) - } else { - error - }) - })?; - if projection.map.file_size != size || projection.map.file_content_id != digest { +) -> Result, Response> +{ + if f.size == 0 { return Err(mst2_error_response(SnapshotError::new( - SnapshotErrorCode::IntegrityError, - "cached projection disagrees with the fixed verified fact", + SnapshotErrorCode::ScopeInvalid, + "empty files have no chunk map", ))); } - Ok(projection) + let source = ChunkMapSource::from_fact(f.fact.clone(), &f.oid).map_err(mst2_error_response)?; + let storage = handler.get_context(); + let repository = storage.chunk_maps().await.map_err(mst2_error_response)?; + let objects = &storage.git_service.obj_storage; + if let Some(map) = repository + .read(&source, objects) + .await + .map_err(mst2_error_response)? + { + return Ok(map); + } + let flight = crate::ceres::snapshot::chunk_map_gate::InstallFlight::acquire( + repository + .source_identity(&source) + .map_err(mst2_error_response)?, + ) + .map_err(mst2_error_response)?; + let _gate = flight.lock().await.map_err(mst2_error_response)?; + if let Some(map) = repository + .read(&source, objects) + .await + .map_err(mst2_error_response)? + { + return Ok(map); + } + let verified = + VerifiedSourceChunkMap::verify(handler, source.clone(), repository.memory_budget()) + .await + .map_err(|error| { + mst2_error_response(if error.code == SnapshotErrorCode::DigestMismatch { + SnapshotError::new( + SnapshotErrorCode::IntegrityError, + "fixed blob digest disagrees with its verified fact", + ) + } else { + error + }) + })?; + repository + .install(verified, objects) + .await + .map_err(mst2_error_response)?; + repository + .read(&source, objects) + .await + .map_err(mst2_error_response)? + .ok_or_else(|| mst2_error_response(internal("installed chunk map source is missing"))) } #[allow(clippy::result_large_err)] @@ -539,10 +562,18 @@ pub(super) async fn chunk_map( q.expected_digest.as_deref(), ) .await?; - // Canonical v3 uses a closed top-level envelope and a nested map - // descriptor. Keeping the map under `map` is part of profile selection; - // the client rejects the legacy flat shape once canonical capabilities - // have been advertised. + let wire_bound = q + .path + .len() + .checked_mul(6) + .and_then(|n| n.checked_add(4096)) + .ok_or_else(|| mst2_error_response(internal("chunk map response bound overflow")))?; + let memory = reserve_response( + wire_bound + .checked_mul(2) + .ok_or_else(|| mst2_error_response(internal("chunk map response credit overflow")))?, + ) + .map_err(mst2_error_response)?; let body = json!({ "snapshot_id": snapshot_id, "path": q.path, @@ -557,7 +588,9 @@ pub(super) async fn chunk_map( "map_id": format!("sha256:{}", hex_of(&proj.map_id)), }, }); - Ok(Json(body).into_response()) + guarded_map_json_response(&state, &ctx, &body, memory) + .await + .map_err(mst2_error_response) } #[allow(clippy::result_large_err)] @@ -569,19 +602,13 @@ pub(super) async fn chunk_map_pages( ensure(&state)?; let ctx = super::request_context(&state, &snapshot_id).map_err(mst2_error_response)?; validate_scope_relative_path(&q.path).map_err(mst2_error_response)?; - // Canonical v3 uses `map_id` + `page_index`; retain parsing of the old - // names only while the legacy client is still present in this checkout. - let page_index: u64 = match (q.page_index.as_deref(), q.page.as_deref()) { - (Some(_), Some(_)) => { - return Err(mst2_error_response(SnapshotError::new( - SnapshotErrorCode::ScopeInvalid, - "page_index and page are mutually exclusive", - ))); - } - (Some(s), None) => parse_decimal_count(s, "page_index").map_err(mst2_error_response)?, - (None, Some(s)) => parse_decimal_count(s, "page").map_err(mst2_error_response)?, - (None, None) => 0, - }; + let page_index = q + .page_index + .as_deref() + .map(|s| parse_decimal_count(s, "page_index")) + .transpose() + .map_err(mst2_error_response)? + .unwrap_or(0); if q.page_index.is_some() != q.map_id.is_some() { return Err(mst2_error_response(SnapshotError::new( SnapshotErrorCode::ScopeInvalid, @@ -611,9 +638,19 @@ pub(super) async fn chunk_map_pages( ))); } } - let (leaf, proof) = proj - .leaf_and_proof(page_index) + let storage = handler.get_context(); + let page = storage + .chunk_maps() + .await + .map_err(mst2_error_response)? + .selected_page(&proj, page_index) + .await .map_err(mst2_error_response)?; + // Reserve both JSON values and the encoded wire buffer before either + // allocation. The authenticated leaf/proof retain their own lease. + let memory = reserve_response(64 * 1024).map_err(mst2_error_response)?; + let leaf = &page.leaf; + let proof = &page.proof; let leaf_bytes = leaf .encode() .map_err(|e| mst2_error_response(internal(format!("chunk leaf encode failed: {e}"))))?; @@ -639,7 +676,84 @@ pub(super) async fn chunk_map_pages( "leaf_base64": base64_of(&leaf_bytes), "proof": proof_json, }); - Ok(Json(body).into_response()) + guarded_map_json_response(&state, &ctx, &body, memory) + .await + .map_err(mst2_error_response) +} + +fn map_json_bytes( + value: &serde_json::Value, + memory: MemoryLease, +) -> Result { + let mut bytes = Vec::new(); + bytes + .try_reserve_exact(memory.bytes / 2) + .map_err(|_| internal("chunk map JSON allocation failed"))?; + let limit = memory.bytes / 2; + let mut writer = BoundedMapJsonWriter { + bytes: &mut bytes, + limit, + }; + serde_json::to_writer(&mut writer, value) + .map_err(|_| internal("chunk map JSON encoding exceeds its owned credit"))?; + if bytes.capacity() > memory.bytes { + return Err(internal( + "chunk map JSON allocation exceeds its owned credit", + )); + } + Ok(bytes::Bytes::from_owner(BudgetedFrame { + bytes, + lease: std::sync::Arc::new(memory), + })) +} + +struct BoundedMapJsonWriter<'a> { + bytes: &'a mut Vec, + limit: usize, +} + +impl std::io::Write for BoundedMapJsonWriter<'_> { + fn write(&mut self, input: &[u8]) -> std::io::Result { + if input.len() > self.limit - self.bytes.len() { + return Err(std::io::Error::other( + "chunk map JSON exceeds its wire bound", + )); + } + self.bytes.extend_from_slice(input); + Ok(input.len()) + } + + fn flush(&mut self) -> std::io::Result<()> { + Ok(()) + } +} + +async fn guarded_map_json_response( + state: &crate::api::MonoApiServiceState, + context: &crate::ceres::snapshot::runtime::SnapshotContext, + value: &serde_json::Value, + memory: MemoryLease, +) -> Result { + let bytes = map_json_bytes(value, memory)?; + let headers = super::REQUEST_HEADERS.try_with(Clone::clone).map_err(|_| { + SnapshotError::new( + SnapshotErrorCode::Unauthenticated, + "request authentication context missing", + ) + })?; + super::revalidate_access(state, context, &headers).await?; + let state = state.clone(); + let context = context.clone(); + let stream = futures::stream::once(async move { + super::revalidate_access(&state, &context, &headers).await?; + Ok::<_, SnapshotError>(bytes) + }); + Response::builder() + .header("content-type", "application/json") + .header("cache-control", "private, no-cache, no-transform") + .header("vary", "Authorization, Accept") + .body(axum::body::Body::from_stream(stream)) + .map_err(|_| internal("chunk map JSON response build failed")) } #[derive(Deserialize, Debug)] @@ -663,7 +777,8 @@ const CHUNKS_MAX_ITEMS: usize = 128; const CHUNKS_TOTAL_MAX: u64 = 128 * 1024 * 1024; struct Planned { - projection: std::sync::Arc, + projection: std::sync::Arc, + page: std::sync::Arc, oid: String, index: u64, } @@ -743,8 +858,9 @@ async fn read_chunk_range( } raw.extend_from_slice(&bytes); } - projection - .verify_chunk(planned.index, &raw) + planned + .page + .verify_chunk(&projection.map, planned.index, &raw) .map_err(mst2_error_response)?; Ok(raw) } @@ -784,7 +900,6 @@ pub(super) async fn chunks( let mut resolved: Vec = Vec::new(); let mut units: Vec<(String, u64)> = Vec::new(); let mut distinct: std::collections::HashMap<[u8; 32], u64> = std::collections::HashMap::new(); - let mut projection_bytes = 0usize; let mut logical_bytes = 0u64; for item in &req.items { validate_scope_relative_path(&item.path).map_err(mst2_error_response)?; @@ -836,16 +951,10 @@ pub(super) async fn chunks( } } std::collections::hash_map::Entry::Vacant(entry) => { - let bytes = projection_reservation_bytes(file.size).map_err(mst2_error_response)?; - projection_bytes = projection_bytes.checked_add(bytes).ok_or_else(|| { - mst2_error_response(internal("projection batch memory overflow")) - })?; - if projection_bytes > PROJECTION_LIVE_BYTES { - return Err(mst2_error_response(SnapshotError::new( - SnapshotErrorCode::LimitExceeded, - "chunk batch exceeds its live projection memory budget", - ))); - } + // Check the profile before any body I/O. Cold construction + // owns its credits and is dropped after durable installation; + // warm requests retain only descriptors and selected pages. + map_build_reservation_bytes(file.size).map_err(mst2_error_response)?; entry.insert(file.size); } } @@ -860,8 +969,19 @@ pub(super) async fn chunks( .and_then(|bytes| bytes.checked_add(req.items.len() * 1024 + 1024)) .ok_or_else(|| mst2_error_response(internal("chunk response memory overflow")))?; let response_memory = reserve_response(response_bytes).map_err(mst2_error_response)?; + let mut maps = std::collections::HashMap::new(); + let mut pages = std::collections::HashMap::new(); for item in resolved { - let proj = project_resolved(handler.as_ref(), &item.file).await?; + let source = ChunkMapSource::from_fact(item.file.fact.clone(), &item.file.oid) + .map_err(mst2_error_response)?; + let source_key = source.canonical_bytes().map_err(mst2_error_response)?; + let proj = if let Some(map) = maps.get(&source_key) { + std::sync::Arc::clone(map) + } else { + let map = project_resolved(handler.as_ref(), &item.file).await?; + maps.insert(source_key, std::sync::Arc::clone(&map)); + map + }; let want_map = format!("sha256:{}", hex_of(&proj.map_id)); if want_map != item.map_id { return Err(mst2_error_response(SnapshotError::new( @@ -869,8 +989,27 @@ pub(super) async fn chunks( "map_id does not bind to the fixed file", ))); } + let page_key = ( + proj.source_id(), + item.index / mst2_codec::chunkmap::CHUNKS_PER_PAGE as u64, + ); + let page = if let Some(page) = pages.get(&page_key) { + std::sync::Arc::clone(page) + } else { + let storage = handler.get_context(); + let page = storage + .chunk_maps() + .await + .map_err(mst2_error_response)? + .selected_page(&proj, page_key.1) + .await + .map_err(mst2_error_response)?; + pages.insert(page_key, std::sync::Arc::clone(&page)); + page + }; planned.push(Planned { projection: proj, + page, oid: item.file.oid, index: item.index, }); @@ -883,14 +1022,7 @@ pub(super) async fn chunks( let _work_memory = reserve_range_work().map_err(mst2_error_response)?; // The source is the current request's fixed OID, never a cached // handler/backend/credential from a different scope. - let bytes = if p.projection.has_inline_bytes() { - p.projection - .chunk_bytes(p.index) - .map_err(mst2_error_response)? - .to_vec() - } else { - read_chunk_range(handler.as_ref(), p).await? - }; + let bytes = read_chunk_range(handler.as_ref(), p).await?; let frame = stream .chunk( p.projection.map_id, diff --git a/src/api/router/snapshot_content_tests.rs b/src/api/router/snapshot_content_tests.rs index bb58d788..4399e692 100644 --- a/src/api/router/snapshot_content_tests.rs +++ b/src/api/router/snapshot_content_tests.rs @@ -1,7 +1,7 @@ use std::{ sync::{ Arc, - atomic::{AtomicUsize, Ordering}, + atomic::{AtomicBool, AtomicUsize, Ordering}, }, time::Duration, }; @@ -85,6 +85,9 @@ mod bounded_objects; #[path = "snapshot_chunks_bounded_tests.rs"] mod bounded_chunks; +#[path = "snapshot_persisted_chunk_map_tests.rs"] +mod persisted_chunk_maps; + #[path = "snapshot_session_tests.rs"] mod durable_sessions; @@ -100,6 +103,8 @@ mod generation_qualified_fixture; #[path = "snapshot_install_capability_fixture.rs"] mod install_capability_fixture; +type ReceiptWriteHold = (Arc, Arc); + #[derive(Default)] struct ReadCounts { whole: AtomicUsize, @@ -107,6 +112,14 @@ struct ReadCounts { bytes: AtomicUsize, object_fault: std::sync::Mutex>, chunk_faults: std::sync::Mutex>, + receipt_reads: AtomicUsize, + receipt_writes: AtomicUsize, + receipt_write_fail_after_create: AtomicBool, + receipt_read_failure: AtomicBool, + receipt_read_corruption: std::sync::Mutex>, + receipt_read_meta_size: std::sync::atomic::AtomicI64, + receipt_read_late_error: AtomicBool, + receipt_write_holds: std::sync::Mutex>, } impl ReadCounts { @@ -114,6 +127,8 @@ impl ReadCounts { self.whole.store(0, Ordering::SeqCst); self.range.store(0, Ordering::SeqCst); self.bytes.store(0, Ordering::SeqCst); + self.receipt_reads.store(0, Ordering::SeqCst); + self.receipt_writes.store(0, Ordering::SeqCst); } fn assert(&self, whole: usize, bytes: usize) { @@ -130,6 +145,36 @@ struct CountingStorage { #[async_trait::async_trait] impl MegaObjectStorage for CountingStorage { + async fn put_metadata_atomic_create( + &self, + key: &ObjectKey, + bytes: Bytes, + meta: ObjectMeta, + ) -> OrbitResult<()> { + self.inner + .inner + .put_metadata_atomic_create(key, bytes, meta) + .await?; + if key.namespace == ObjectNamespace::ChunkMapReceipt { + self.counts.receipt_writes.fetch_add(1, Ordering::SeqCst); + let holds = self.counts.receipt_write_holds.lock().unwrap().clone(); + if let Some((entered, release)) = holds { + entered.notify_one(); + release.notified().await; + } + if self + .counts + .receipt_write_fail_after_create + .swap(false, Ordering::SeqCst) + { + return Err(crate::orbit_api::error::IoOrbitError::Other( + "injected post-create receipt failure".into(), + )); + } + } + Ok(()) + } + async fn put_stream( &self, key: &ObjectKey, @@ -140,6 +185,39 @@ impl MegaObjectStorage for CountingStorage { } async fn get_stream(&self, key: &ObjectKey) -> OrbitResult<(ObjectByteStream, ObjectMeta)> { + if key.namespace == ObjectNamespace::ChunkMapReceipt { + self.counts.receipt_reads.fetch_add(1, Ordering::SeqCst); + if self.counts.receipt_read_failure.load(Ordering::SeqCst) { + return Err( + crate::orbit_api::error::IoOrbitError::object_store_not_found( + key.default_sharding(), + ), + ); + } + let bad = self.counts.receipt_read_corruption.lock().unwrap().clone(); + if let Some(bytes) = bad { + let declared = self.counts.receipt_read_meta_size.load(Ordering::SeqCst); + let size = if declared > 0 { + declared + } else { + bytes.len() as i64 + }; + let mut parts = vec![Ok(bytes)]; + if self.counts.receipt_read_late_error.load(Ordering::SeqCst) { + parts.push(Err(std::io::Error::other( + "injected late receipt read error", + ))); + } + return Ok(( + Box::pin(futures::stream::iter(parts)), + ObjectMeta { + size, + ..Default::default() + }, + )); + } + return self.inner.inner.get_stream(key).await; + } self.counts.whole.fetch_add(1, Ordering::SeqCst); let chunk_fault = self .counts @@ -220,10 +298,21 @@ impl MegaObjectStorage for CountingStorage { (Box::pin(stream) as ObjectByteStream, meta) })); } - self.inner + let result = self + .inner .inner .get_range_stream_exact(key, start, end) - .await + .await?; + Ok(result.map(|(stream, meta)| { + let counts = self.counts.clone(); + let stream = stream.map(move |part| { + if let Ok(bytes) = &part { + counts.bytes.fetch_add(bytes.len(), Ordering::SeqCst); + } + part + }); + (Box::pin(stream) as ObjectByteStream, meta) + })) } async fn exists(&self, key: &ObjectKey) -> OrbitResult { @@ -500,11 +589,16 @@ impl Fixture { .save_object_from_raw(Bytes::copy_from_slice(raw)) .await .unwrap(); - project_items.push(item( - TreeItemMode::Blob, - ObjectHash::from_hex_for_kind(HashKind::Sha1, &oid).unwrap(), - name, - )); + let mut oid = ObjectHash::from_hex_for_kind(HashKind::Sha1, &oid).unwrap(); + let mut mode = TreeItemMode::Blob; + let components: Vec<_> = name.split('/').collect(); + for index in (1..components.len()).rev() { + let child = tree(vec![item(mode, oid, components[index])]); + oid = child.id; + mode = TreeItemMode::Tree; + extra_trees.push(child); + } + project_items.push(item(mode, oid, components[0])); } let project = tree(project_items); let old_tip = Commit::from_tree_id_with_kind( @@ -814,7 +908,7 @@ async fn mst2_fixed_head_uses_verified_facts_without_body_reads_and_preserves_ra } #[tokio::test] -async fn mst2_fixed_warm_map_leaf_and_chunk_aliases_skip_body_reads() { +async fn mst2_fixed_warm_map_and_leaf_aliases_skip_body_reads_and_chunks_use_current_ranges() { let fixture = Fixture::new().await; let initial = fixture.map("/file").await; fixture.counts.assert(1, fixture.raw.len()); @@ -823,18 +917,6 @@ async fn mst2_fixed_warm_map_leaf_and_chunk_aliases_skip_body_reads() { assert_eq!(initial["map"]["chunk_count"], "2"); let map_id = initial["map"]["map_id"].as_str().unwrap(); fixture.counts.reset(); - fixture - .state - .storage - .git_service - .obj_storage - .inner - .delete(&ObjectKey { - namespace: ObjectNamespace::Git, - key: fixture.oid.clone(), - }) - .await - .unwrap(); for path in ["/file", "/alias", "/executable", "/nested/file"] { let cached = fixture.map(path).await; assert_eq!(cached["path"], path); @@ -902,8 +984,43 @@ async fn mst2_fixed_warm_map_leaf_and_chunk_aliases_skip_body_reads() { <[u8; 32]>::from(Sha256::digest(body.as_bytes())) ); } - fixture.counts.assert(0, 0); + assert_eq!(fixture.counts.whole.load(Ordering::SeqCst), 0); + assert_eq!(fixture.counts.range.load(Ordering::SeqCst), 2); + assert_eq!( + fixture.counts.bytes.load(Ordering::SeqCst), + fixture.raw.len() + ); + fixture.counts.reset(); } + fixture + .state + .storage + .git_service + .obj_storage + .inner + .delete(&ObjectKey { + namespace: ObjectNamespace::Git, + key: fixture.oid.clone(), + }) + .await + .unwrap(); + assert_eq!(fixture.map("/alias").await["map"], initial["map"]); + fixture.counts.assert(0, 0); + error( + fixture + .send( + "POST", + "chunks", + Body::from(fixture.chunk_body("/alias", map_id, "0").to_string()), + ) + .await, + 503, + "OBJECT_UNAVAILABLE", + false, + ) + .await; + assert_eq!(fixture.counts.whole.load(Ordering::SeqCst), 0); + assert_eq!(fixture.counts.range.load(Ordering::SeqCst), 1); } #[tokio::test] diff --git a/src/api/router/snapshot_objects_bounded_tests.rs b/src/api/router/snapshot_objects_bounded_tests.rs index d001de12..b5e0d303 100644 --- a/src/api/router/snapshot_objects_bounded_tests.rs +++ b/src/api/router/snapshot_objects_bounded_tests.rs @@ -7,11 +7,11 @@ use super::*; #[derive(Clone)] pub(super) struct StreamFault { pub(super) oid: String, - kind: FaultKind, + pub(super) kind: FaultKind, } #[derive(Clone)] -enum FaultKind { +pub(super) enum FaultKind { Parts(Vec), LateError(Bytes), Oversized(Bytes, Arc), @@ -21,6 +21,11 @@ enum FaultKind { release: Arc, drops: Arc, }, + HeldThenError { + entered: Arc, + release: Arc, + drops: Arc, + }, } struct DropCount(Arc); @@ -34,6 +39,24 @@ impl Drop for DropCount { impl StreamFault { pub(super) fn stream(self) -> ObjectByteStream { match self.kind { + FaultKind::HeldThenError { + entered, + release, + drops, + } => Box::pin(futures::stream::unfold( + (true, entered, release, DropCount(drops)), + |(first, entered, release, owner)| async move { + if !first { + return None; + } + entered.notify_one(); + release.notified().await; + Some(( + Err(io::Error::other("held source read failed")), + (false, entered, release, owner), + )) + }, + )), FaultKind::Parts(parts) => Box::pin(futures::stream::iter(parts.into_iter().map(Ok))), FaultKind::LateError(raw) => Box::pin(futures::stream::iter([ Ok(raw), diff --git a/src/api/router/snapshot_persisted_chunk_map_tests.rs b/src/api/router/snapshot_persisted_chunk_map_tests.rs new file mode 100644 index 00000000..6b261115 --- /dev/null +++ b/src/api/router/snapshot_persisted_chunk_map_tests.rs @@ -0,0 +1,1303 @@ +use sea_orm::{DatabaseConnection, DbBackend, IsolationLevel, Statement, TransactionTrait}; +use tokio::{sync::Notify, time::timeout}; + +use super::*; +use crate::{ + ceres::snapshot::{ + chunks::{ChunkMapSource, ChunkProjection}, + content_budget::MemoryBudget, + }, + jupiter::storage::native_chunk_map::PostgresChunkMapRepository, +}; + +fn statement(sql: &str, values: [sea_orm::Value; N]) -> Statement { + Statement::from_sql_and_values(DbBackend::Postgres, sql, values) +} + +async fn count(db: &DatabaseConnection, table: &str) -> i64 { + db.query_one_raw(statement( + &format!("SELECT count(*) AS count FROM {table}"), + [], + )) + .await + .unwrap() + .unwrap() + .try_get("", "count") + .unwrap() +} + +async fn reconstructed(fixture: &Fixture) -> MonoApiServiceState { + let mut config = (*fixture.state.storage.config()).clone(); + config.database.max_connection = 1; + config.database.min_connection = 1; + let config = Arc::new(config); + let connection = crate::jupiter::storage::init::postgres_connection(&config.database) + .await + .unwrap(); + let storage = crate::jupiter::storage::Storage::new_with_connection( + config, + Arc::new(connection), + fixture.state.storage.git_service.obj_storage.clone(), + ) + .await + .unwrap(); + MonoApiServiceState { + storage, + git_object_cache: Arc::new(GitObjectCache { + connection: fixture.state.git_object_cache.connection.clone(), + prefix: uuid::Uuid::new_v4().to_string(), + }), + ..fixture.state.clone() + } +} + +fn router(state: &MonoApiServiceState) -> Router { + Router::new().nest("/api/v2", routers(state.clone()).with_state(state.clone())) +} + +#[tokio::test] +async fn persisted_map_rebuilt_actual_http_uses_canonical_pages_and_only_requested_raw_range() { + let fixture = Fixture::new_with_pg_config(true).await; + let original = fixture.map("/file").await; + fixture.counts.assert(1, fixture.raw.len()); + assert_eq!(fixture.counts.receipt_writes.load(Ordering::SeqCst), 1); + let oracle = ChunkProjection::build(fixture.digest, fixture.raw.clone()).unwrap(); + assert_eq!( + original["map"]["map_id"], + format!("sha256:{}", hex_of(&oracle.map_id)) + ); + let state = reconstructed(&fixture).await; + assert!(state.storage.native_chunk_maps.get().is_none()); + let app = router(&state); + fixture.counts.reset(); + let map = success_json( + app.clone() + .oneshot(fixture.request("GET", "chunk-map?path=/alias", Body::empty())) + .await + .unwrap(), + ) + .await; + assert_eq!(map["map"], original["map"]); + let map_id = map["map"]["map_id"].as_str().unwrap(); + let page = success_json( + app.clone() + .oneshot(fixture.request( + "GET", + &format!("chunk-map/pages?path=/alias&map_id={map_id}&page_index=0"), + Body::empty(), + )) + .await + .unwrap(), + ) + .await; + let bytes = STANDARD + .decode(page["leaf_base64"].as_str().unwrap()) + .unwrap(); + let (leaf, proof) = oracle.leaf_and_proof(0).unwrap(); + assert_eq!(bytes, leaf.encode().unwrap()); + assert!(proof.is_empty()); + assert_eq!(page["proof"], json!([])); + fixture.counts.assert(0, 0); + assert!(fixture.counts.receipt_reads.load(Ordering::SeqCst) >= 2); + assert_eq!(fixture.counts.receipt_writes.load(Ordering::SeqCst), 0); + let body = fixture.chunk_body("/alias", map_id, "1").to_string(); + let response = app + .oneshot(fixture.request("POST", "chunks", Body::from(body.clone()))) + .await + .unwrap(); + assert_eq!(response.status(), 200); + let wire = to_bytes(response.into_body(), 2 * 1024 * 1024) + .await + .unwrap(); + let frames = parse_stream(&wire).unwrap(); + let [Frame::Chunk(chunk), Frame::End(end)] = frames.as_slice() else { + panic!("expected exact CHUNK and END"); + }; + assert_eq!(chunk.chunk_bytes, fixture.raw[CHUNK_SIZE as usize..]); + assert_eq!(chunk.chunk_index, 1); + assert_eq!(chunk.file_content_id, fixture.digest); + assert_eq!(chunk.map_id, oracle.map_id); + assert_eq!(end.request_item_count, 1); + assert_eq!(end.unique_unit_count, 1); + assert_eq!(end.logical_bytes, 113); + assert_eq!( + end.request_body_sha256, + <[u8; 32]>::from(Sha256::digest(body.as_bytes())) + ); + assert_eq!(fixture.counts.whole.load(Ordering::SeqCst), 0); + assert_eq!(fixture.counts.range.load(Ordering::SeqCst), 1); + assert_eq!(fixture.counts.bytes.load(Ordering::SeqCst), 113); +} + +#[tokio::test] +async fn persisted_map_current_fact_tuple_receipt_and_leaf_corruption_fail_closed_without_fallback() +{ + let fixture = Fixture::new().await; + let map = fixture.map("/file").await; + let original = fixture.fact().await; + let map_id = map["map"]["map_id"].as_str().unwrap(); + fixture.counts.reset(); + for case in 0..3 { + let mut fact = original.clone(); + match case { + 0 => fact.id += 1_000_000, + 1 => fact.created_at += chrono::Duration::seconds(1), + _ => fact.raw_sha256[0] ^= 1, + } + fixture.replace_fact(fact).await; + error( + fixture + .send("GET", "chunk-map?path=/file", Body::empty()) + .await, + 502, + "INTEGRITY_ERROR", + false, + ) + .await; + fixture.counts.assert(0, 0); + } + fixture.replace_fact(original).await; + fixture + .counts + .receipt_read_failure + .store(true, Ordering::SeqCst); + error( + fixture + .send("GET", "chunk-map?path=/file", Body::empty()) + .await, + 502, + "INTEGRITY_ERROR", + false, + ) + .await; + fixture + .counts + .receipt_read_failure + .store(false, Ordering::SeqCst); + *fixture.counts.receipt_read_corruption.lock().unwrap() = + Some(Bytes::from_static(b"forged DB proof")); + error( + fixture + .send("GET", "chunk-map?path=/file", Body::empty()) + .await, + 502, + "INTEGRITY_ERROR", + false, + ) + .await; + *fixture.counts.receipt_read_corruption.lock().unwrap() = None; + let mono = fixture.state.storage.mono_storage(); + let db = mono.get_connection(); + let leaf: Vec = db + .query_one_raw(statement( + "SELECT payload FROM mst2_chunk_map_leaf WHERE map_id=$1 AND page_index=0", + [hex::decode(map_id.trim_start_matches("sha256:")) + .unwrap() + .into()], + )) + .await + .unwrap() + .unwrap() + .try_get("", "payload") + .unwrap(); + db.execute_unprepared("ALTER TABLE mst2_chunk_map_leaf DISABLE TRIGGER USER") + .await + .unwrap(); + let mut bad = leaf.clone(); + bad[16] ^= 1; + db.execute_raw(statement( + "UPDATE mst2_chunk_map_leaf SET payload=$1", + [bad.into()], + )) + .await + .unwrap(); + db.execute_unprepared("ALTER TABLE mst2_chunk_map_leaf ENABLE TRIGGER USER") + .await + .unwrap(); + for (method, suffix, body) in [ + ( + "GET", + format!("chunk-map/pages?path=/file&map_id={map_id}&page_index=0"), + Body::empty(), + ), + ( + "POST", + "chunks".to_string(), + Body::from(fixture.chunk_body("/file", map_id, "0").to_string()), + ), + ] { + error( + fixture.send(method, &suffix, body).await, + 502, + "INTEGRITY_ERROR", + false, + ) + .await; + } + fixture.counts.assert(0, 0); + db.execute_unprepared("ALTER TABLE mst2_chunk_map_leaf DISABLE TRIGGER USER") + .await + .unwrap(); + db.execute_raw(statement( + "UPDATE mst2_chunk_map_leaf SET payload=$1", + [leaf.into()], + )) + .await + .unwrap(); + db.execute_unprepared("ALTER TABLE mst2_chunk_map_leaf ENABLE TRIGGER USER") + .await + .unwrap(); + assert_eq!(fixture.map("/file").await["map"], map["map"]); + fixture.counts.assert(0, 0); +} + +#[tokio::test] +async fn ordinary_db_dml_cannot_forge_full_body_admission_or_mutate_admitted_indexes() { + let fixture = Fixture::new().await; + let source = ChunkMapSource::from_fact(fixture.fact().await, &fixture.oid).unwrap(); + let mono = fixture.state.storage.mono_storage(); + let db = mono.get_connection(); + let repository = PostgresChunkMapRepository::new(db.clone()).await.unwrap(); + let scope = repository.test_primary_scope(); + let source_bytes = source.canonical_bytes().unwrap(); + let leaf = ChunkLeaf { + page_index: 0, + chunk_sha256: vec![[9; 32]; 2], + }; + let root = leaf.leaf_hash().unwrap(); + let map = mst2_codec::chunkmap::ChunkMap::new(fixture.digest, fixture.raw.len() as u64, root) + .unwrap(); + let mut receipt = b"MST2-CHUNK-MAP-RECEIPT\0".to_vec(); + receipt.extend_from_slice(&(scope.len() as u32).to_be_bytes()); + receipt.extend_from_slice(scope); + receipt.extend_from_slice(&(source_bytes.len() as u32).to_be_bytes()); + receipt.extend_from_slice(&source_bytes); + let source_id: [u8; 32] = Sha256::digest(&receipt).into(); + receipt.extend_from_slice(&map.encode()); + let txn = db + .begin_with_config(Some(IsolationLevel::ReadCommitted), None) + .await + .unwrap(); + txn.execute_raw(statement( + "INSERT INTO mst2_chunk_map(map_id,descriptor,page_count,pages_root) VALUES($1,$2,1,$3)", + [ + map.map_id().to_vec().into(), + map.encode().into(), + root.to_vec().into(), + ], + )) + .await + .unwrap(); + txn.execute_raw(statement( + "INSERT INTO mst2_chunk_map_leaf(map_id,page_index,payload) VALUES($1,0,$2)", + [map.map_id().to_vec().into(), leaf.encode().unwrap().into()], + )) + .await + .unwrap(); + txn.execute_raw(statement( + "INSERT INTO mst2_chunk_map_node(map_id,first_page,page_count,digest) VALUES($1,0,1,$2)", + [map.map_id().to_vec().into(), root.to_vec().into()], + )) + .await + .unwrap(); + txn.execute_raw(statement("INSERT INTO mst2_chunk_map_source(storage_domain,git_oid,object_kind,fact_id,source_id,source_bytes,primary_scope,map_id,receipt_digest) VALUES('git',$1,'blob',$2,$3,$4,$5,$6,$7)", [fixture.oid.clone().into(),source.fact().id.into(),source_id.to_vec().into(),source_bytes.into(),scope.to_vec().into(),map.map_id().to_vec().into(),Sha256::digest(&receipt).to_vec().into()])).await.unwrap(); + txn.commit().await.unwrap(); + // All row checks and even a self-computed receipt digest pass. They + // still cannot create the trusted writer's independent object receipt. + fixture.counts.reset(); + error( + fixture + .send("GET", "chunk-map?path=/file", Body::empty()) + .await, + 502, + "INTEGRITY_ERROR", + false, + ) + .await; + fixture.counts.assert(0, 0); + assert_eq!(fixture.counts.receipt_writes.load(Ordering::SeqCst), 0); + for table in [ + "mst2_chunk_map", + "mst2_chunk_map_leaf", + "mst2_chunk_map_node", + "mst2_chunk_map_source", + ] { + assert!( + db.execute_unprepared(&format!("DELETE FROM {table}")) + .await + .is_err() + ); + assert!( + db.execute_unprepared(&format!("TRUNCATE {table} CASCADE")) + .await + .is_err() + ); + } + assert!( + db.execute_unprepared("UPDATE mst2_chunk_map_source SET fact_id=fact_id") + .await + .is_err() + ); + assert!(db.execute_raw(statement("INSERT INTO mst2_chunk_map_node(map_id,first_page,page_count,digest) VALUES($1,1,1,$2)", [map.map_id().to_vec().into(),root.to_vec().into()])).await.is_err()); +} + +#[tokio::test] +async fn receipt_orphan_failure_and_cancelled_install_release_owned_credit_and_replay_atomically() { + for cancel in [false, true] { + let fixture = Fixture::new().await; + let mono = fixture.state.storage.mono_storage(); + let budget = MemoryBudget::new(8 * 1024 * 1024); + let repository = PostgresChunkMapRepository::new(mono.get_connection().clone()) + .await + .unwrap() + .with_test_budget(budget.clone()); + assert!( + fixture + .state + .storage + .native_chunk_maps + .set(repository) + .is_ok() + ); + if cancel { + let entered = Arc::new(Notify::new()); + let release = Arc::new(Notify::new()); + *fixture.counts.receipt_write_holds.lock().unwrap() = Some((entered.clone(), release)); + let app = fixture.app.clone(); + let request = fixture.request("GET", "chunk-map?path=/file", Body::empty()); + let task = tokio::spawn(async move { app.oneshot(request).await }); + timeout(Duration::from_secs(10), entered.notified()) + .await + .unwrap(); + assert!(budget.used() > 4 * 1024 * 1024); + assert_eq!( + count(mono.get_connection(), "mst2_chunk_map_source").await, + 0 + ); + task.abort(); + assert!(task.await.err().unwrap().is_cancelled()); + *fixture.counts.receipt_write_holds.lock().unwrap() = None; + } else { + fixture + .counts + .receipt_write_fail_after_create + .store(true, Ordering::SeqCst); + error( + fixture + .send("GET", "chunk-map?path=/file", Body::empty()) + .await, + 503, + "TEMPORARY_UNAVAILABLE", + true, + ) + .await; + fixture + .counts + .receipt_write_fail_after_create + .store(false, Ordering::SeqCst); + } + assert_eq!(budget.used(), 0); + fixture.counts.assert(1, fixture.raw.len()); + assert_eq!(fixture.counts.receipt_writes.load(Ordering::SeqCst), 1); + assert_eq!(count(mono.get_connection(), "mst2_chunk_map").await, 0); + fixture.counts.reset(); + fixture.map("/file").await; + fixture.counts.assert(1, fixture.raw.len()); + assert_eq!(fixture.counts.receipt_writes.load(Ordering::SeqCst), 1); + assert_eq!(count(mono.get_connection(), "mst2_chunk_map").await, 1); + assert_eq!( + count(mono.get_connection(), "mst2_chunk_map_source").await, + 1 + ); + assert_eq!(count(mono.get_connection(), "mst2_chunk_map_leaf").await, 1); + assert_eq!(count(mono.get_connection(), "mst2_chunk_map_node").await, 1); + fixture.counts.reset(); + fixture.map("/file").await; + fixture.counts.assert(0, 0); + assert_eq!(fixture.counts.receipt_writes.load(Ordering::SeqCst), 0); + assert_eq!(budget.used(), 0); + } +} + +#[tokio::test] +async fn map_json_transport_bytes_keep_owned_credit_until_the_last_clone_drops() { + let budget = MemoryBudget::new(4096); + let bytes = super::super::map_json_bytes( + &json!({"map_id":"sha256:owned"}), + budget.reserve(4096).unwrap(), + ) + .unwrap(); + let mut stream = Body::from(bytes).into_data_stream(); + let transport = stream.next().await.unwrap().unwrap(); + let clone = transport.clone(); + drop(transport); + drop(stream); + assert_eq!(budget.used(), 4096); + assert!(budget.reserve(1).is_err()); + drop(clone); + assert_eq!(budget.used(), 0); + assert!(budget.reserve(4096).is_ok()); +} + +#[test] +fn json_wire_limit_rejects_growth_and_refunds_credit() { + let budget = MemoryBudget::new(4096); + let value = json!({"path":"\u{1}".repeat(1000)}); + let error = super::super::map_json_bytes(&value, budget.reserve(4096).unwrap()) + .err() + .unwrap(); + assert_eq!(error.code, SnapshotErrorCode::Internal); + assert_eq!(budget.used(), 0); +} + +async fn observer(fixture: &Fixture) -> crate::ceres::snapshot::chunk_map_gate::InstallFlight { + let repository = fixture.state.storage.chunk_maps().await.unwrap(); + let source = ChunkMapSource::from_fact(fixture.fact().await, &fixture.oid).unwrap(); + crate::ceres::snapshot::chunk_map_gate::InstallFlight::acquire( + repository.source_identity(&source).unwrap(), + ) + .unwrap() +} + +async fn wait_owners( + flight: &crate::ceres::snapshot::chunk_map_gate::InstallFlight, + owners: usize, +) { + timeout(Duration::from_secs(10), async { + loop { + if flight.test_owner_count() == owners { + break; + } + tokio::task::yield_now().await; + } + }) + .await + .expect("actual HTTP callers did not reach the same-source install gate"); +} + +fn held_leader(fixture: &Fixture) -> (Arc, Arc, tokio::task::JoinHandle) { + let entered = Arc::new(Notify::new()); + let release = Arc::new(Notify::new()); + *fixture.counts.receipt_write_holds.lock().unwrap() = Some((entered.clone(), release.clone())); + let app = fixture.app.clone(); + let request = fixture.request("GET", "chunk-map?path=/file", Body::empty()); + let task = tokio::spawn(async move { app.oneshot(request).await.unwrap() }); + (entered, release, task) +} + +#[tokio::test] +async fn same_source_cold_actual_http_callers_share_one_full_pass_and_each_recheck_their_receipt() { + let fixture = Fixture::new().await; + let (entered, release, leader) = held_leader(&fixture); + timeout(Duration::from_secs(10), entered.notified()) + .await + .unwrap(); + let flight = observer(&fixture).await; + let mut joined = Vec::new(); + for _ in 0..6 { + let app = fixture.app.clone(); + let request = fixture.request("GET", "chunk-map?path=/alias", Body::empty()); + joined.push(tokio::spawn( + async move { app.oneshot(request).await.unwrap() }, + )); + } + let other_counts = Arc::new(ReadCounts::default()); + other_counts + .receipt_read_failure + .store(true, Ordering::SeqCst); + let mut other_state = fixture.state.clone(); + other_state.storage.git_service = GitService { + obj_storage: MegaObjectStorageWrapper::new(Arc::new(CountingStorage { + inner: fixture.state.storage.git_service.obj_storage.clone(), + counts: other_counts.clone(), + })), + }; + let other_app = router(&other_state); + let request = fixture.request("GET", "chunk-map?path=/file", Body::empty()); + let rejected = tokio::spawn(async move { other_app.oneshot(request).await.unwrap() }); + wait_owners(&flight, 9).await; // Leader, seven callers and this observer. + fixture.counts.assert(1, fixture.raw.len()); + release.notify_one(); + let expected = success_json(leader.await.unwrap()).await; + for task in joined { + assert_eq!( + success_json(task.await.unwrap()).await["map"], + expected["map"] + ); + } + error(rejected.await.unwrap(), 502, "INTEGRITY_ERROR", false).await; + assert_eq!(other_counts.receipt_reads.load(Ordering::SeqCst), 1); + assert_eq!(other_counts.whole.load(Ordering::SeqCst), 0); + fixture.counts.assert(1, fixture.raw.len()); + assert_eq!(fixture.counts.receipt_writes.load(Ordering::SeqCst), 1); + assert_eq!(fixture.counts.receipt_reads.load(Ordering::SeqCst), 8); + wait_owners(&flight, 1).await; +} + +#[tokio::test] +async fn failed_or_cancelled_leader_and_cancelled_waiter_leave_actual_http_retry_capacity() { + for mode in 0..3 { + let fixture = Fixture::new().await; + let budget = MemoryBudget::new(8 * 1024 * 1024); + let repository = PostgresChunkMapRepository::new( + fixture + .state + .storage + .mono_storage() + .get_connection() + .clone(), + ) + .await + .unwrap() + .with_test_budget(budget.clone()); + assert!( + fixture + .state + .storage + .native_chunk_maps + .set(repository) + .is_ok() + ); + let (entered, release, leader) = held_leader(&fixture); + timeout(Duration::from_secs(10), entered.notified()) + .await + .unwrap(); + let flight = observer(&fixture).await; + let app = fixture.app.clone(); + let request = fixture.request("GET", "chunk-map?path=/alias", Body::empty()); + let waiter = tokio::spawn(async move { app.oneshot(request).await.unwrap() }); + wait_owners(&flight, 3).await; + assert!(budget.used() > 4 * 1024 * 1024); + *fixture.counts.receipt_write_holds.lock().unwrap() = None; + match mode { + 0 => { + fixture + .counts + .receipt_write_fail_after_create + .store(true, Ordering::SeqCst); + release.notify_one(); + error(leader.await.unwrap(), 503, "TEMPORARY_UNAVAILABLE", true).await; + success_json(waiter.await.unwrap()).await; + } + 1 => { + leader.abort(); + assert!(leader.await.err().unwrap().is_cancelled()); + success_json(waiter.await.unwrap()).await; + } + _ => { + waiter.abort(); + assert!(waiter.await.err().unwrap().is_cancelled()); + wait_owners(&flight, 2).await; + release.notify_one(); + success_json(leader.await.unwrap()).await; + } + } + wait_owners(&flight, 1).await; + drop(flight); + assert_eq!(budget.used(), 0); + let passes = if mode == 2 { 1 } else { 2 }; + fixture.counts.assert(passes, passes * fixture.raw.len()); + assert_eq!(fixture.counts.receipt_writes.load(Ordering::SeqCst), passes); + fixture.counts.reset(); + fixture.map("/file").await; + fixture.counts.assert(0, 0); + assert_eq!(budget.used(), 0); + } +} + +#[tokio::test] +async fn actual_http_long_escaped_legal_path_has_bounded_owned_json_and_empty_files_have_no_map() { + let component = "\u{1}".repeat(255); + let mut components = vec![component; 15]; + components.push("\u{1}".repeat(239)); + let name = components.join("/"); + let path = format!("/{name}"); + assert_eq!(path.len(), 4080); + crate::ceres::snapshot::view::validate_scope_relative_path(&path).unwrap(); + let fixture = Fixture::new_with_pg_config_directories_and_objects( + false, + 0, + &[(name, b"escaped path body".to_vec())], + ) + .await; + let encoded = url::form_urlencoded::byte_serialize(path.as_bytes()).collect::(); + for whole in [1, 0] { + fixture.counts.reset(); + let response = fixture + .send("GET", &format!("chunk-map?path={encoded}"), Body::empty()) + .await; + assert_eq!(response.status(), 200); + assert_eq!( + response.headers()["cache-control"], + "private, no-cache, no-transform" + ); + let bytes = to_bytes(response.into_body(), 64 * 1024).await.unwrap(); + assert!(bytes.len() > 4 * 1024); + let value: Value = serde_json::from_slice(&bytes).unwrap(); + assert_eq!(value["path"], path); + assert_eq!(value["map"]["file_size"], "17"); + fixture + .counts + .assert(whole, if whole == 0 { 0 } else { 17 }); + } + fixture.counts.reset(); + for suffix in ["chunk-map?path=/empty", "chunk-map/pages?path=/empty"] { + error( + fixture.send("GET", suffix, Body::empty()).await, + 400, + "SCOPE_INVALID", + false, + ) + .await; + } + fixture.counts.assert(0, 0); +} + +#[tokio::test] +async fn actual_http_and_repository_initialization_ignore_poisoned_temp_fact_scope_and_map_shadows() +{ + let fixture = Fixture::new_with_pg_config(true).await; + let expected = fixture.map("/file").await; + let state = reconstructed(&fixture).await; + assert!(state.storage.native_chunk_maps.get().is_none()); + let mono = state.storage.mono_storage(); + let db = mono.get_connection(); + let schema = fixture + ._schema + .as_ref() + .unwrap() + .schema() + .replace('"', "\"\""); + let tables = [ + "mst2_verified_object", + "mst2_metadata_storage_scope", + "mst2_chunk_map", + "mst2_chunk_map_source", + "mst2_chunk_map_leaf", + "mst2_chunk_map_node", + ]; + for table in tables { + db.execute_unprepared(&format!("CREATE TEMP TABLE {table}(LIKE \"{schema}\".{table}); INSERT INTO pg_temp.{table} SELECT * FROM \"{schema}\".{table}")).await.unwrap(); + } + db.execute_unprepared("UPDATE pg_temp.mst2_verified_object SET raw_sha256=decode(repeat('09',32),'hex'); UPDATE pg_temp.mst2_metadata_storage_scope SET storage_uuid='temp-poison'; UPDATE pg_temp.mst2_chunk_map SET descriptor=decode('00','hex'); UPDATE pg_temp.mst2_chunk_map_leaf SET payload=decode('00','hex'); UPDATE pg_temp.mst2_chunk_map_node SET digest=decode(repeat('09',32),'hex')").await.unwrap(); + fixture.counts.reset(); + let app = router(&state); + let map = success_json( + app.clone() + .oneshot(fixture.request("GET", "chunk-map?path=/file", Body::empty())) + .await + .unwrap(), + ) + .await; + assert_eq!(map, expected); + let map_id = map["map"]["map_id"].as_str().unwrap(); + let page = success_json( + app.clone() + .oneshot(fixture.request( + "GET", + &format!("chunk-map/pages?path=/file&map_id={map_id}&page_index=0"), + Body::empty(), + )) + .await + .unwrap(), + ) + .await; + let oracle = ChunkProjection::build(fixture.digest, fixture.raw.clone()).unwrap(); + assert_eq!( + STANDARD + .decode(page["leaf_base64"].as_str().unwrap()) + .unwrap(), + oracle.leaf_and_proof(0).unwrap().0.encode().unwrap() + ); + let body = fixture.chunk_body("/file", map_id, "1").to_string(); + let response = app + .clone() + .oneshot(fixture.request("POST", "chunks", Body::from(body))) + .await + .unwrap(); + assert_eq!(response.status(), 200); + let wire = to_bytes(response.into_body(), 2 * 1024 * 1024) + .await + .unwrap(); + let frames = parse_stream(&wire).unwrap(); + let [Frame::Chunk(chunk), Frame::End(end)] = frames.as_slice() else { + panic!("expected CHUNK and END"); + }; + assert_eq!(chunk.chunk_bytes, fixture.raw[CHUNK_SIZE as usize..]); + assert_eq!(end.logical_bytes, 113); + assert_eq!(fixture.counts.whole.load(Ordering::SeqCst), 0); + assert_eq!(fixture.counts.range.load(Ordering::SeqCst), 1); + assert_eq!(fixture.counts.bytes.load(Ordering::SeqCst), 113); + assert_eq!(fixture.counts.receipt_writes.load(Ordering::SeqCst), 0); + // Real primary changes still fail despite an apparently healthy shadow. + let actual = fixture.state.storage.mono_storage(); + let source = ChunkMapSource::from_fact(fixture.fact().await, &fixture.oid).unwrap(); + let repository = state.storage.chunk_maps().await.unwrap(); + actual.get_connection().execute_unprepared("ALTER TABLE mst2_metadata_storage_scope DISABLE TRIGGER USER; UPDATE mst2_metadata_storage_scope SET storage_uuid='changed-real-primary'; ALTER TABLE mst2_metadata_storage_scope ENABLE TRIGGER USER").await.unwrap(); + assert_eq!( + repository + .read(&source, &state.storage.git_service.obj_storage) + .await + .err() + .unwrap() + .code, + crate::ceres::snapshot::error::SnapshotErrorCode::IntegrityError + ); + error( + app.oneshot(fixture.request("GET", "chunk-map?path=/file", Body::empty())) + .await + .unwrap(), + 502, + "INTEGRITY_ERROR", + false, + ) + .await; +} + +pub(super) async fn assert_three_page_proofs_and_selected_sibling_faults( + fixture: &Fixture, + map_id: &str, + digest: [u8; 32], + pattern: &[u8], +) { + use mst2_codec::chunkmap::{ChunkMap, ProofSide, leaf_proof, merkle_root}; + let full: [u8; 32] = Sha256::digest(pattern).into(); + let final_chunk: [u8; 32] = Sha256::digest(&pattern[..7]).into(); + let leaves = [ + ChunkLeaf { + page_index: 0, + chunk_sha256: vec![full; 256], + }, + ChunkLeaf { + page_index: 1, + chunk_sha256: vec![full; 256], + }, + ChunkLeaf { + page_index: 2, + chunk_sha256: vec![final_chunk], + }, + ]; + let hashes: Vec<_> = leaves + .iter() + .map(|leaf| leaf.leaf_hash().unwrap()) + .collect(); + let root = merkle_root(&hashes).unwrap(); + let oracle = ChunkMap::new(digest, 512 * CHUNK_SIZE as u64 + 7, root).unwrap(); + assert_eq!(map_id, format!("sha256:{}", hex_of(&oracle.map_id()))); + for index in 0..3 { + let page = success_json( + fixture + .send( + "GET", + &format!("chunk-map/pages?path=/file&map_id={map_id}&page_index={index}"), + Body::empty(), + ) + .await, + ) + .await; + assert_eq!( + STANDARD + .decode(page["leaf_base64"].as_str().unwrap()) + .unwrap(), + leaves[index as usize].encode().unwrap() + ); + let proof = leaf_proof(&hashes, index).unwrap(); + let expected: Vec<_> = proof.iter().map(|step| json!({ + "side": if step.side == ProofSide::Left { "left" } else { "right" }, + "sibling_pages": step.sibling_pages.to_string(), "digest": format!("sha256:{}", hex_of(&step.digest)), + })).collect(); + assert_eq!(page["proof"], json!(expected)); + verify_leaf( + 3, + index, + leaves[index as usize].leaf_hash().unwrap(), + &proof, + root, + ) + .unwrap(); + } + fixture.counts.assert(0, 0); + let mono = fixture.state.storage.mono_storage(); + let db = mono.get_connection(); + let id = oracle.map_id().to_vec(); + for missing in [false, true] { + db.execute_unprepared("ALTER TABLE mst2_chunk_map_node DISABLE TRIGGER USER") + .await + .unwrap(); + if missing { + db.execute_raw(statement( + "DELETE FROM mst2_chunk_map_node WHERE map_id=$1 AND first_page=2 AND page_count=1", + [id.clone().into()], + )) + .await + .unwrap(); + } else { + db.execute_raw(statement("UPDATE mst2_chunk_map_node SET digest=$2 WHERE map_id=$1 AND first_page=2 AND page_count=1", [id.clone().into(), vec![9; 32].into()])).await.unwrap(); + } + db.execute_unprepared("ALTER TABLE mst2_chunk_map_node ENABLE TRIGGER USER") + .await + .unwrap(); + for (method, suffix, body) in [ + ( + "GET", + format!("chunk-map/pages?path=/file&map_id={map_id}&page_index=0"), + Body::empty(), + ), + ( + "POST", + "chunks".to_string(), + Body::from(fixture.chunk_body("/file", map_id, "0").to_string()), + ), + ] { + error( + fixture.send(method, &suffix, body).await, + 502, + "INTEGRITY_ERROR", + false, + ) + .await; + } + fixture.counts.assert(0, 0); + // Page 2 proves itself through [0,2); the unrelated damaged node is + // absent from its selected SQL and does not turn into a full-map scan. + let page = success_json( + fixture + .send( + "GET", + &format!("chunk-map/pages?path=/file&map_id={map_id}&page_index=2"), + Body::empty(), + ) + .await, + ) + .await; + assert_eq!( + STANDARD + .decode(page["leaf_base64"].as_str().unwrap()) + .unwrap(), + leaves[2].encode().unwrap() + ); + db.execute_unprepared("ALTER TABLE mst2_chunk_map_node DISABLE TRIGGER USER") + .await + .unwrap(); + if missing { + db.execute_raw(statement("INSERT INTO mst2_chunk_map_node(map_id,first_page,page_count,digest) VALUES($1,2,1,$2)", [id.clone().into(), hashes[2].to_vec().into()])).await.unwrap(); + } else { + db.execute_raw(statement("UPDATE mst2_chunk_map_node SET digest=$2 WHERE map_id=$1 AND first_page=2 AND page_count=1", [id.clone().into(), hashes[2].to_vec().into()])).await.unwrap(); + } + db.execute_unprepared("ALTER TABLE mst2_chunk_map_node ENABLE TRIGGER USER") + .await + .unwrap(); + } + fixture.counts.assert(0, 0); +} + +#[tokio::test] +async fn failed_digest_failed_body_and_cancelled_cold_producer_allow_the_joined_current_source_to_retry() + { + use super::bounded_objects::{FaultKind, StreamFault}; + for mode in 0..3 { + let fixture = Fixture::new().await; + let budget = MemoryBudget::new(8 * 1024 * 1024); + let repository = PostgresChunkMapRepository::new( + fixture + .state + .storage + .mono_storage() + .get_connection() + .clone(), + ) + .await + .unwrap() + .with_test_budget(budget.clone()); + assert!( + fixture + .state + .storage + .native_chunk_maps + .set(repository) + .is_ok() + ); + let entered = Arc::new(Notify::new()); + let release = Arc::new(Notify::new()); + let drops = Arc::new(AtomicUsize::new(0)); + let kind = if mode == 1 { + FaultKind::HeldThenError { + entered: entered.clone(), + release: release.clone(), + drops: drops.clone(), + } + } else { + let mut raw = fixture.raw.clone(); + if mode == 0 { + raw[0] ^= 1; + } + FaultKind::Held { + raw: Bytes::from(raw), + entered: entered.clone(), + release: release.clone(), + drops: drops.clone(), + } + }; + *fixture.counts.object_fault.lock().unwrap() = Some(StreamFault { + oid: fixture.oid.clone(), + kind, + }); + let app = fixture.app.clone(); + let request = fixture.request("GET", "chunk-map?path=/file", Body::empty()); + let leader = tokio::spawn(async move { app.oneshot(request).await.unwrap() }); + timeout(Duration::from_secs(10), entered.notified()) + .await + .unwrap(); + let flight = observer(&fixture).await; + let app = fixture.app.clone(); + let request = fixture.request("GET", "chunk-map?path=/alias", Body::empty()); + let waiter = tokio::spawn(async move { app.oneshot(request).await.unwrap() }); + wait_owners(&flight, 3).await; + assert_eq!(fixture.counts.receipt_writes.load(Ordering::SeqCst), 0); + assert!(budget.used() > 4 * 1024 * 1024); + *fixture.counts.object_fault.lock().unwrap() = None; + if mode == 2 { + leader.abort(); + assert!(leader.await.err().unwrap().is_cancelled()); + } else { + release.notify_one(); + error( + leader.await.unwrap(), + if mode == 0 { 502 } else { 503 }, + if mode == 0 { + "INTEGRITY_ERROR" + } else { + "OBJECT_UNAVAILABLE" + }, + false, + ) + .await; + } + let map = success_json(waiter.await.unwrap()).await; + assert_eq!( + map["map"]["file_content_id"], + format!("sha256:{}", hex_of(&fixture.digest)) + ); + assert_eq!(drops.load(Ordering::SeqCst), 1); + assert_eq!(fixture.counts.whole.load(Ordering::SeqCst), 2); + assert_eq!(fixture.counts.receipt_writes.load(Ordering::SeqCst), 1); + let failed_bytes = match mode { + 0 => fixture.raw.len(), + 1 => 0, + _ => 1, + }; + assert_eq!( + fixture.counts.bytes.load(Ordering::SeqCst), + fixture.raw.len() + failed_bytes + ); + wait_owners(&flight, 1).await; + drop(flight); + assert_eq!(budget.used(), 0); + fixture.counts.reset(); + fixture.map("/file").await; + fixture.counts.assert(0, 0); + } +} + +#[tokio::test] +async fn different_current_sources_enter_cold_installations_independently() { + let fixture = Fixture::new_with_pg_config_directories_and_objects( + false, + 0, + &[("other".into(), b"independent source".to_vec())], + ) + .await; + let (entered, release, leader) = held_leader(&fixture); + timeout(Duration::from_secs(10), entered.notified()) + .await + .unwrap(); + let app = fixture.app.clone(); + let request = fixture.request("GET", "chunk-map?path=/other", Body::empty()); + let other = tokio::spawn(async move { app.oneshot(request).await.unwrap() }); + timeout(Duration::from_secs(10), entered.notified()) + .await + .unwrap(); + assert_eq!(fixture.counts.whole.load(Ordering::SeqCst), 2); + assert_eq!(fixture.counts.receipt_writes.load(Ordering::SeqCst), 2); + release.notify_waiters(); + let map = success_json(leader.await.unwrap()).await; + let second = success_json(other.await.unwrap()).await; + assert_ne!(map["map"]["map_id"], second["map"]["map_id"]); + fixture.counts.assert(2, fixture.raw.len() + 18); +} + +#[tokio::test] +async fn persisted_descriptors_and_authenticated_pages_charge_until_the_last_live_reader_drops() { + let fixture = Fixture::new().await; + fixture.map("/file").await; + fixture.counts.reset(); + let budget = MemoryBudget::new(96 * 1024); + let repository = PostgresChunkMapRepository::new( + fixture + .state + .storage + .mono_storage() + .get_connection() + .clone(), + ) + .await + .unwrap() + .with_test_budget(budget.clone()); + let source = ChunkMapSource::from_fact(fixture.fact().await, &fixture.oid).unwrap(); + let map = repository + .read(&source, &fixture.state.storage.git_service.obj_storage) + .await + .unwrap() + .unwrap(); + let page = repository.selected_page(&map, 0).await.unwrap(); + let other_map = map.clone(); + let other_page = page.clone(); + drop(map); + drop(page); + assert_eq!(budget.used(), 96 * 1024); + assert_eq!( + budget.reserve(1).err().unwrap().code, + SnapshotErrorCode::TemporaryUnavailable + ); + other_page + .verify_chunk(&other_map.map, 1, &fixture.raw[CHUNK_SIZE as usize..]) + .unwrap(); + drop(other_map); + assert_eq!(budget.used(), 64 * 1024); + drop(other_page); + assert_eq!(budget.used(), 0); + let replay = repository + .read(&source, &fixture.state.storage.git_service.obj_storage) + .await + .unwrap() + .unwrap(); + assert_eq!(budget.used(), 32 * 1024); + drop(replay); + assert_eq!(budget.used(), 0); + fixture.counts.assert(0, 0); +} + +#[tokio::test] +async fn actual_install_sql_failure_rolls_back_every_index_and_exact_retry_earns_admission_again() { + let fixture = Fixture::new().await; + let mono = fixture.state.storage.mono_storage(); + let db = mono.get_connection(); + let budget = MemoryBudget::new(8 * 1024 * 1024); + let repository = PostgresChunkMapRepository::new(db.clone()) + .await + .unwrap() + .with_test_budget(budget.clone()); + assert!( + fixture + .state + .storage + .native_chunk_maps + .set(repository) + .is_ok() + ); + db.execute_unprepared("CREATE FUNCTION chunk_map_install_test_failure() RETURNS trigger LANGUAGE plpgsql AS $test$ BEGIN RAISE EXCEPTION 'injected source-row failure after complete index insertion'; END $test$; CREATE TRIGGER chunk_map_install_test_failure BEFORE INSERT ON mst2_chunk_map_source FOR EACH ROW EXECUTE FUNCTION chunk_map_install_test_failure()").await.unwrap(); + error( + fixture + .send("GET", "chunk-map?path=/file", Body::empty()) + .await, + 503, + "TEMPORARY_UNAVAILABLE", + true, + ) + .await; + fixture.counts.assert(1, fixture.raw.len()); + assert_eq!(fixture.counts.receipt_writes.load(Ordering::SeqCst), 1); + assert_eq!(budget.used(), 0); + for table in [ + "mst2_chunk_map", + "mst2_chunk_map_leaf", + "mst2_chunk_map_node", + "mst2_chunk_map_source", + ] { + assert_eq!(count(db, table).await, 0); + } + db.execute_unprepared("DROP TRIGGER chunk_map_install_test_failure ON mst2_chunk_map_source; DROP FUNCTION chunk_map_install_test_failure()").await.unwrap(); + fixture.counts.reset(); + let map = fixture.map("/file").await; + fixture.counts.assert(1, fixture.raw.len()); + assert_eq!(fixture.counts.receipt_writes.load(Ordering::SeqCst), 1); + assert_eq!(budget.used(), 0); + for table in [ + "mst2_chunk_map", + "mst2_chunk_map_leaf", + "mst2_chunk_map_node", + "mst2_chunk_map_source", + ] { + assert_eq!(count(db, table).await, 1); + } + fixture.counts.reset(); + assert_eq!(fixture.map("/alias").await["map"], map["map"]); + fixture.counts.assert(0, 0); + assert_eq!(fixture.counts.receipt_writes.load(Ordering::SeqCst), 0); +} + +#[tokio::test] +async fn ordinary_incomplete_source_dml_cannot_commit_an_admitted_partial_map() { + let fixture = Fixture::new().await; + let mono = fixture.state.storage.mono_storage(); + let db = mono.get_connection(); + let source = ChunkMapSource::from_fact(fixture.fact().await, &fixture.oid).unwrap(); + let repository = PostgresChunkMapRepository::new(db.clone()).await.unwrap(); + let map = ChunkProjection::build(fixture.digest, fixture.raw.clone()).unwrap(); + let scope = repository.test_primary_scope(); + let txn = db + .begin_with_config(Some(IsolationLevel::ReadCommitted), None) + .await + .unwrap(); + txn.execute_raw(statement( + "INSERT INTO mst2_chunk_map(map_id,descriptor,page_count,pages_root) VALUES($1,$2,$3,$4)", + [ + map.map_id.to_vec().into(), + map.map.encode().into(), + (map.map.page_count as i32).into(), + map.map.pages_root.to_vec().into(), + ], + )) + .await + .unwrap(); + txn.execute_raw(statement("INSERT INTO mst2_chunk_map_source(storage_domain,git_oid,object_kind,fact_id,source_id,source_bytes,primary_scope,map_id,receipt_digest) VALUES('git',$1,'blob',$2,$3,$4,$5,$6,$7)", [fixture.oid.clone().into(), source.fact().id.into(), repository.source_identity(&source).unwrap().to_vec().into(), source.canonical_bytes().unwrap().into(), scope.to_vec().into(), map.map_id.to_vec().into(), vec![9;32].into()])).await.unwrap(); + assert!(txn.commit().await.is_err()); + for table in [ + "mst2_chunk_map", + "mst2_chunk_map_leaf", + "mst2_chunk_map_node", + "mst2_chunk_map_source", + ] { + assert_eq!(count(db, table).await, 0); + } + fixture.counts.assert(0, 0); +} + +#[tokio::test] +async fn actual_map_and_page_json_revalidate_lease_before_the_first_transport_poll() { + for page in [false, true] { + let fixture = Fixture::new().await; + let map = fixture.map("/file").await; + let suffix = if page { + format!( + "chunk-map/pages?path=/file&map_id={}&page_index=0", + map["map"]["map_id"].as_str().unwrap() + ) + } else { + "chunk-map?path=/file".into() + }; + fixture.counts.reset(); + let response = fixture.send("GET", &suffix, Body::empty()).await; + assert_eq!(response.status(), 200); + let revoked = fixture + .app + .clone() + .oneshot( + Request::builder() + .method("DELETE") + .uri(format!("/api/v2/snapshots/leases/{}", fixture.lease)) + .header("authorization", format!("Bearer {TOKEN}")) + .body(Body::empty()) + .unwrap(), + ) + .await + .unwrap(); + assert_eq!(revoked.status(), 200); + let mut stream = response.into_body().into_data_stream(); + assert!(stream.next().await.unwrap().is_err()); + assert!(stream.next().await.is_none()); + fixture.counts.assert(0, 0); + } +} + +#[tokio::test] +async fn oversized_current_fact_digest_fails_before_source_or_receipt_io_and_recovers_canonical_bytes() + { + let fixture = Fixture::new().await; + let expected = fixture.map("/file").await; + let original = fixture.fact().await; + let mono = fixture.state.storage.mono_storage(); + mono.get_connection().execute_raw(statement("UPDATE mst2_verified_object SET raw_sha256=decode(repeat('ab',1048576),'hex') WHERE id=$1", [original.id.into()])).await.unwrap(); + fixture.counts.reset(); + error( + fixture + .send("GET", "chunk-map?path=/file", Body::empty()) + .await, + 502, + "INTEGRITY_ERROR", + false, + ) + .await; + fixture.counts.assert(0, 0); + assert_eq!(fixture.counts.receipt_reads.load(Ordering::SeqCst), 0); + fixture.replace_fact(original).await; + assert_eq!(fixture.map("/file").await, expected); + fixture.counts.assert(0, 0); +} + +#[tokio::test] +async fn warm_admission_requires_receipt_exact_bytes_size_and_final_eof_without_fallback() { + let fixture = Fixture::new().await; + let map = fixture.map("/file").await; + let repository = fixture.state.storage.chunk_maps().await.unwrap(); + let source = ChunkMapSource::from_fact(fixture.fact().await, &fixture.oid).unwrap(); + let key = ObjectKey { + namespace: ObjectNamespace::ChunkMapReceipt, + key: hex_of(&repository.source_identity(&source).unwrap()), + }; + let (mut stream, meta) = fixture + .state + .storage + .git_service + .obj_storage + .inner + .get_stream(&key) + .await + .unwrap(); + let mut original = Vec::new(); + while let Some(part) = stream.next().await { + original.extend_from_slice(&part.unwrap()); + } + assert_eq!(original.len() as i64, meta.size); + let mut wrong = original.clone(); + wrong[0] ^= 1; + let mut long = original.clone(); + long.push(0); + let cases = [ + original[..original.len() - 1].to_vec(), + long, + wrong, + original.clone(), + ]; + fixture + .counts + .receipt_read_meta_size + .store(meta.size, Ordering::SeqCst); + for (index, bytes) in cases.into_iter().enumerate() { + fixture.counts.reset(); + *fixture.counts.receipt_read_corruption.lock().unwrap() = Some(Bytes::from(bytes)); + fixture + .counts + .receipt_read_late_error + .store(index == 3, Ordering::SeqCst); + error( + fixture + .send("GET", "chunk-map?path=/file", Body::empty()) + .await, + 502, + "INTEGRITY_ERROR", + false, + ) + .await; + fixture.counts.assert(0, 0); + assert_eq!(fixture.counts.receipt_reads.load(Ordering::SeqCst), 1); + assert_eq!(fixture.counts.receipt_writes.load(Ordering::SeqCst), 0); + } + *fixture.counts.receipt_read_corruption.lock().unwrap() = None; + fixture + .counts + .receipt_read_late_error + .store(false, Ordering::SeqCst); + assert_eq!(fixture.map("/file").await, map); + fixture.counts.assert(0, 0); +} diff --git a/src/ceres/snapshot/chunk_map_gate.rs b/src/ceres/snapshot/chunk_map_gate.rs new file mode 100644 index 00000000..5c56a504 --- /dev/null +++ b/src/ceres/snapshot/chunk_map_gate.rs @@ -0,0 +1,185 @@ +//! Bounded exact-source install flights. Completed data is always re-read +//! from the requesting source's trusted receipt, never retained in a flight. + +use std::{ + collections::HashMap, + sync::{Arc, Mutex, OnceLock, Weak}, +}; + +use super::error::{SnapshotError, SnapshotErrorCode}; + +const MAX_INSTALL_FLIGHTS: usize = 128; +type SourceGate = tokio::sync::Mutex<()>; + +#[derive(Default)] +struct Registry { + entries: HashMap<[u8; 32], Weak>, +} + +pub(crate) struct InstallFlight { + key: [u8; 32], + gate: Option>, + registry: Arc>, +} + +impl InstallFlight { + pub(crate) fn acquire(key: [u8; 32]) -> Result { + static REGISTRY: OnceLock>> = OnceLock::new(); + Self::from_registry(REGISTRY.get_or_init(|| Arc::default()).clone(), key) + } + + fn from_registry(registry: Arc>, key: [u8; 32]) -> Result { + let gate = { + let mut state = registry.lock().map_err(|_| internal())?; + if let Some(gate) = state.entries.get(&key).and_then(Weak::upgrade) { + gate + } else { + state.entries.remove(&key); + if state.entries.len() >= MAX_INSTALL_FLIGHTS { + return Err(SnapshotError::new( + SnapshotErrorCode::LimitExceeded, + "too many distinct source map installations in flight", + )); + } + let gate = Arc::new(SourceGate::new(())); + state.entries.insert(key, Arc::downgrade(&gate)); + gate + } + }; + Ok(Self { + key, + gate: Some(gate), + registry, + }) + } + + pub(crate) async fn lock(&self) -> Result, SnapshotError> { + Ok(self.gate.as_ref().ok_or_else(internal)?.lock().await) + } + + #[cfg(test)] + pub(crate) fn test_owner_count(&self) -> usize { + self.gate.as_ref().map_or(0, Arc::strong_count) + } +} + +impl Drop for InstallFlight { + fn drop(&mut self) { + let Some(gate) = self.gate.take() else { + return; + }; + if let Ok(mut state) = self.registry.lock() { + if Arc::strong_count(&gate) == 1 + && state + .entries + .get(&self.key) + .is_some_and(|entry| entry.ptr_eq(&Arc::downgrade(&gate))) + { + state.entries.remove(&self.key); + } + // Owners must release their strong reference while the registry + // is locked, so concurrent final drops cannot leave a stale slot. + drop(gate); + } + } +} + +fn internal() -> SnapshotError { + SnapshotError::new( + SnapshotErrorCode::Internal, + "source map flight registry is unavailable", + ) +} + +#[cfg(test)] +mod tests { + use super::*; + + #[tokio::test] + async fn exact_source_gate_blocks_only_same_source_and_returns_capacity_after_last_owner() { + let registry = Arc::new(Mutex::new(Registry::default())); + let first = InstallFlight::from_registry(registry.clone(), [1; 32]).unwrap(); + let lock = first.lock().await.unwrap(); + let same = InstallFlight::from_registry(registry.clone(), [1; 32]).unwrap(); + assert!(same.gate.as_ref().unwrap().try_lock().is_err()); + let other = InstallFlight::from_registry(registry.clone(), [2; 32]).unwrap(); + assert!(other.gate.as_ref().unwrap().try_lock().is_ok()); + drop(lock); + drop(first); + assert!(same.gate.as_ref().unwrap().try_lock().is_ok()); + assert_eq!(registry.lock().unwrap().entries.len(), 2); + drop(same); + drop(other); + assert!(registry.lock().unwrap().entries.is_empty()); + let mut held = Vec::new(); + for i in 0..MAX_INSTALL_FLIGHTS { + let mut key = [0; 32]; + key[..8].copy_from_slice(&(i as u64).to_le_bytes()); + held.push(InstallFlight::from_registry(registry.clone(), key).unwrap()); + } + assert_eq!( + InstallFlight::from_registry(registry.clone(), [255; 32]) + .err() + .unwrap() + .code, + SnapshotErrorCode::LimitExceeded + ); + let joined_at_capacity = InstallFlight::from_registry(registry.clone(), [0; 32]).unwrap(); + assert_eq!(joined_at_capacity.test_owner_count(), 2); + assert_eq!(registry.lock().unwrap().entries.len(), MAX_INSTALL_FLIGHTS); + drop(joined_at_capacity); + drop(held); + assert!(registry.lock().unwrap().entries.is_empty()); + assert!(InstallFlight::from_registry(registry, [255; 32]).is_ok()); + } + + #[test] + fn concurrent_last_owners_release_registry_capacity() { + for _ in 0..64 { + let registry = Arc::new(Mutex::new(Registry::default())); + let first = InstallFlight::from_registry(registry.clone(), [1; 32]).unwrap(); + let second = InstallFlight::from_registry(registry.clone(), [1; 32]).unwrap(); + let barrier = Arc::new(std::sync::Barrier::new(2)); + std::thread::scope(|scope| { + let other = barrier.clone(); + scope.spawn(move || { + other.wait(); + drop(first); + }); + scope.spawn(move || { + barrier.wait(); + drop(second); + }); + }); + assert!(registry.lock().unwrap().entries.is_empty()); + } + } + + #[test] + fn dropping_an_old_claim_does_not_remove_a_replacement_gate() { + let registry = Arc::new(Mutex::new(Registry::default())); + let old = InstallFlight::from_registry(registry.clone(), [1; 32]).unwrap(); + let replacement = Arc::new(SourceGate::new(())); + registry + .lock() + .unwrap() + .entries + .insert([1; 32], Arc::downgrade(&replacement)); + drop(old); + assert_eq!(registry.lock().unwrap().entries.len(), 1); + let current = InstallFlight::from_registry(registry.clone(), [1; 32]).unwrap(); + assert!(Arc::ptr_eq(current.gate.as_ref().unwrap(), &replacement)); + drop(replacement); + drop(current); + assert!(registry.lock().unwrap().entries.is_empty()); + registry + .lock() + .unwrap() + .entries + .insert([1; 32], Weak::new()); + let reclaimed = InstallFlight::from_registry(registry.clone(), [1; 32]).unwrap(); + assert_eq!(registry.lock().unwrap().entries.len(), 1); + drop(reclaimed); + assert!(registry.lock().unwrap().entries.is_empty()); + } +} diff --git a/src/ceres/snapshot/chunk_map_index.rs b/src/ceres/snapshot/chunk_map_index.rs new file mode 100644 index 00000000..7e253f9c --- /dev/null +++ b/src/ceres/snapshot/chunk_map_index.rs @@ -0,0 +1,171 @@ +//! Canonical MCL2 Merkle subtrees addressed by their exact leaf interval. + +use mst2_codec::chunkmap::{ProofSide, ProofStep}; +use sha2::{Digest, Sha256}; + +use super::error::{SnapshotError, SnapshotErrorCode}; + +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub(crate) struct ChunkMapNode { + pub start: u64, + pub pages: u64, + pub digest: [u8; 32], +} + +pub(crate) fn indexed_nodes(hashes: &[[u8; 32]]) -> Result, SnapshotError> { + if hashes.is_empty() || hashes.len() > 32_768 { + return Err(integrity("chunk map is outside the indexed page profile")); + } + let mut nodes = Vec::new(); + nodes.try_reserve_exact(hashes.len() * 2 - 1).map_err(|_| { + SnapshotError::new( + SnapshotErrorCode::TemporaryUnavailable, + "chunk map index allocation failed", + ) + })?; + build(hashes, 0, &mut nodes); + Ok(nodes) +} + +fn build(hashes: &[[u8; 32]], start: u64, nodes: &mut Vec) -> [u8; 32] { + let pages = hashes.len() as u64; + let digest = if pages == 1 { + hashes[0] + } else { + let split = split(pages); + let left = build(&hashes[..split as usize], start, nodes); + let right = build(&hashes[split as usize..], start + split, nodes); + let mut hash = Sha256::new(); + hash.update(b"mega.mst2.chunkbranch\0"); + hash.update(split.to_le_bytes()); + hash.update(left); + hash.update((pages - split).to_le_bytes()); + hash.update(right); + hash.finalize().into() + }; + nodes.push(ChunkMapNode { + start, + pages, + digest, + }); + digest +} + +fn split(pages: u64) -> u64 { + 1 << (63 - (pages - 1).leading_zeros()) +} + +/// Only these sibling intervals may be fetched for a selected page. +pub(crate) fn proof_intervals( + pages: u64, + index: u64, +) -> Result, SnapshotError> { + if pages == 0 || pages > 32_768 || index >= pages { + return Err(SnapshotError::new( + SnapshotErrorCode::PathNotFound, + "chunk map page does not exist", + )); + } + let (mut start, mut count) = (0, pages); + let mut intervals = Vec::new(); + while count > 1 { + let left = split(count); + if index < start + left { + intervals.push((ProofSide::Right, start + left, count - left)); + count = left; + } else { + intervals.push((ProofSide::Left, start, left)); + start += left; + count -= left; + } + } + intervals.reverse(); + Ok(intervals) +} + +pub(crate) fn selected_proof( + intervals: &[(ProofSide, u64, u64)], + nodes: &[ChunkMapNode], +) -> Result, SnapshotError> { + if nodes.len() != intervals.len() { + return Err(integrity( + "persisted chunk map proof coverage is incomplete", + )); + } + intervals + .iter() + .map(|&(side, start, pages)| { + let mut matching = nodes + .iter() + .filter(|n| n.start == start && n.pages == pages); + let node = matching + .next() + .ok_or_else(|| integrity("persisted chunk map sibling is missing"))?; + if matching.next().is_some() { + return Err(integrity("persisted chunk map sibling is duplicated")); + } + Ok(ProofStep { + side, + sibling_pages: pages, + digest: node.digest, + }) + }) + .collect() +} + +fn integrity(message: &str) -> SnapshotError { + SnapshotError::new(SnapshotErrorCode::IntegrityError, message) +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn indexed_selected_proofs_match_independent_codec_for_uneven_trees() { + for count in [1, 2, 3, 5, 9, 17, 257, 32_768] { + let hashes: Vec<[u8; 32]> = (0u64..count) + .map(|n| Sha256::digest(n.to_le_bytes()).into()) + .collect(); + let nodes = indexed_nodes(&hashes).unwrap(); + assert_eq!(nodes.len(), hashes.len() * 2 - 1); + let root = mst2_codec::chunkmap::merkle_root(&hashes).unwrap(); + assert_eq!(nodes.last().unwrap().digest, root); + for page in [0, count / 2, count - 1] { + let intervals = proof_intervals(count, page).unwrap(); + assert!(intervals.len() <= 15); + let selected: Vec<_> = nodes + .iter() + .filter(|n| { + intervals + .iter() + .any(|&(_, s, c)| n.start == s && n.pages == c) + }) + .copied() + .collect(); + let proof = selected_proof(&intervals, &selected).unwrap(); + assert_eq!( + proof, + mst2_codec::chunkmap::leaf_proof(&hashes, page).unwrap() + ); + mst2_codec::chunkmap::verify_leaf(count, page, hashes[page as usize], &proof, root) + .unwrap(); + if !selected.is_empty() { + assert!(selected_proof(&intervals, &selected[1..]).is_err()); + let mut bad = proof.clone(); + bad[0].digest[0] ^= 1; + assert!( + mst2_codec::chunkmap::verify_leaf( + count, + page, + hashes[page as usize], + &bad, + root + ) + .is_err() + ); + } + } + } + } +} diff --git a/src/ceres/snapshot/chunk_singleflight_tests.rs b/src/ceres/snapshot/chunk_singleflight_tests.rs deleted file mode 100644 index 78daff83..00000000 --- a/src/ceres/snapshot/chunk_singleflight_tests.rs +++ /dev/null @@ -1,487 +0,0 @@ -use std::{ - sync::{ - Barrier, - atomic::{AtomicUsize, Ordering}, - }, - time::Duration, -}; - -use tokio::{sync::Notify, time::timeout}; - -use super::*; - -fn unique_data(size: usize) -> ([u8; 32], Arc>) { - let seed = uuid::Uuid::new_v4(); - let raw: Vec = (0..size).map(|i| seed.as_bytes()[i % 16]).collect(); - (Sha256::digest(&raw).into(), Arc::new(raw)) -} - -async fn started(signal: &Notify) { - timeout(Duration::from_secs(5), signal.notified()) - .await - .unwrap(); -} - -async fn participants(content_id: [u8; 32], expected: usize) { - timeout(Duration::from_secs(5), async { - loop { - let count = FLIGHTS - .get() - .unwrap() - .lock() - .unwrap() - .entries - .get(&content_id) - .map_or(0, Weak::strong_count); - if count == expected { - break; - } - tokio::task::yield_now().await; - } - }) - .await - .unwrap(); -} - -fn assert_flight_released(content_id: [u8; 32]) { - assert!( - !FLIGHTS - .get() - .unwrap() - .lock() - .unwrap() - .entries - .contains_key(&content_id) - ); -} - -fn assert_uncached(content_id: [u8; 32]) { - assert!( - STAGED - .get() - .unwrap() - .lock() - .unwrap() - .get(content_id) - .is_none() - ); -} - -fn evict(content_id: [u8; 32]) { - let mut cache = STAGED.get().unwrap().lock().unwrap(); - let projection = cache.entries.remove(&content_id).unwrap(); - cache.total_bytes -= projection.retained_bytes(); - let position = cache.order.iter().position(|id| *id == content_id).unwrap(); - cache.order.remove(position); -} - -#[tokio::test] -async fn cold_same_digest_loads_once_and_warm_cache_skips_loader() { - let (content_id, raw) = unique_data(CHUNK_SIZE as usize + 7); - let calls = Arc::new(AtomicUsize::new(0)); - let entered = Arc::new(Notify::new()); - let release = Arc::new(Notify::new()); - let leader = tokio::spawn({ - let raw = raw.clone(); - let calls = calls.clone(); - let entered = entered.clone(); - let release = release.clone(); - async move { - get_or_project(content_id, || async move { - calls.fetch_add(1, Ordering::SeqCst); - entered.notify_one(); - release.notified().await; - Ok(raw.as_ref().clone()) - }) - .await - } - }); - started(&entered).await; - let mut waiters = Vec::new(); - for _ in 0..7 { - let raw = raw.clone(); - let calls = calls.clone(); - waiters.push(tokio::spawn(async move { - get_or_project(content_id, || async move { - calls.fetch_add(1, Ordering::SeqCst); - Ok(raw.as_ref().clone()) - }) - .await - })); - } - participants(content_id, 8).await; - release.notify_one(); - let projection = leader.await.unwrap().unwrap(); - for waiter in waiters { - let shared = waiter.await.unwrap().unwrap(); - assert!(Arc::ptr_eq(&projection, &shared)); - } - assert_eq!(calls.load(Ordering::SeqCst), 1); - assert_eq!(projection.map.file_content_id, content_id); - assert_eq!(projection.map.chunk_count, 2); - assert_eq!( - projection.chunk_bytes(0).unwrap(), - &raw[..CHUNK_SIZE as usize] - ); - assert_eq!( - projection.chunk_bytes(1).unwrap(), - &raw[CHUNK_SIZE as usize..] - ); - let (leaf, proof) = projection.leaf_and_proof(0).unwrap(); - mst2_codec::chunkmap::verify_leaf( - projection.page_count(), - 0, - leaf.leaf_hash().unwrap(), - &proof, - projection.map.pages_root, - ) - .unwrap(); - let warm = get_or_project(content_id, || async { - panic!("warm projection must not invoke its loader"); - }) - .await - .unwrap(); - assert!(Arc::ptr_eq(&projection, &warm)); - assert_flight_released(content_id); -} - -#[tokio::test] -async fn different_digests_enter_their_loaders_independently() { - let mut requests = Vec::new(); - let mut signals = Vec::new(); - for _ in 0..2 { - let (content_id, raw) = unique_data(97); - let entered = Arc::new(Notify::new()); - let release = Arc::new(Notify::new()); - signals.push((content_id, entered.clone(), release.clone())); - requests.push(tokio::spawn(async move { - get_or_project(content_id, || async move { - entered.notify_one(); - release.notified().await; - Ok(raw.as_ref().clone()) - }) - .await - })); - } - for (_, entered, _) in &signals { - started(entered).await; - } - for (_, _, release) in &signals { - release.notify_one(); - } - for request in requests { - request.await.unwrap().unwrap(); - } - for (content_id, _, _) in signals { - assert_flight_released(content_id); - } -} - -async fn failed_leader_then_waiter(wrong_digest: bool) { - let (content_id, raw) = unique_data(103); - let calls = Arc::new(AtomicUsize::new(0)); - let first_entered = Arc::new(Notify::new()); - let first_release = Arc::new(Notify::new()); - let second_entered = Arc::new(Notify::new()); - let second_release = Arc::new(Notify::new()); - let leader = tokio::spawn({ - let calls = calls.clone(); - let entered = first_entered.clone(); - let release = first_release.clone(); - async move { - get_or_project(content_id, || async move { - calls.fetch_add(1, Ordering::SeqCst); - entered.notify_one(); - release.notified().await; - if wrong_digest { - Ok(vec![0; 103]) - } else { - Err(SnapshotError::new( - SnapshotErrorCode::ObjectUnavailable, - "failed test loader", - )) - } - }) - .await - } - }); - started(&first_entered).await; - let waiter = tokio::spawn({ - let raw = raw.clone(); - let calls = calls.clone(); - let entered = second_entered.clone(); - let release = second_release.clone(); - async move { - get_or_project(content_id, || async move { - calls.fetch_add(1, Ordering::SeqCst); - entered.notify_one(); - release.notified().await; - Ok(raw.as_ref().clone()) - }) - .await - } - }); - participants(content_id, 2).await; - first_release.notify_one(); - let failure = leader.await.unwrap().err().unwrap(); - assert_eq!( - failure.code, - if wrong_digest { - SnapshotErrorCode::DigestMismatch - } else { - SnapshotErrorCode::ObjectUnavailable - } - ); - started(&second_entered).await; - assert_eq!(calls.load(Ordering::SeqCst), 2); - assert_uncached(content_id); - second_release.notify_one(); - let projection = waiter.await.unwrap().unwrap(); - assert_eq!(projection.chunk_bytes(0).unwrap(), raw.as_slice()); - assert_flight_released(content_id); -} - -#[tokio::test] -async fn load_failure_is_unpublished_and_waiter_uses_its_loader() { - failed_leader_then_waiter(false).await; -} - -#[tokio::test] -async fn digest_failure_is_unpublished_and_waiter_uses_its_loader() { - failed_leader_then_waiter(true).await; -} - -#[tokio::test] -async fn cancelled_leader_releases_gate_for_waiting_loader() { - let (content_id, raw) = unique_data(113); - let entered = Arc::new(Notify::new()); - let release = Arc::new(Notify::new()); - let takeover = Arc::new(Notify::new()); - let takeover_release = Arc::new(Notify::new()); - let calls = Arc::new(AtomicUsize::new(0)); - let leader = tokio::spawn({ - let raw = raw.clone(); - let calls = calls.clone(); - let entered = entered.clone(); - let release = release.clone(); - async move { - get_or_project(content_id, || async move { - calls.fetch_add(1, Ordering::SeqCst); - entered.notify_one(); - release.notified().await; - Ok(raw.as_ref().clone()) - }) - .await - } - }); - started(&entered).await; - let waiter = tokio::spawn({ - let raw = raw.clone(); - let calls = calls.clone(); - let takeover = takeover.clone(); - let takeover_release = takeover_release.clone(); - async move { - get_or_project(content_id, || async move { - calls.fetch_add(1, Ordering::SeqCst); - takeover.notify_one(); - takeover_release.notified().await; - Ok(raw.as_ref().clone()) - }) - .await - } - }); - participants(content_id, 2).await; - leader.abort(); - assert!(leader.await.err().unwrap().is_cancelled()); - started(&takeover).await; - assert_eq!(calls.load(Ordering::SeqCst), 2); - assert_uncached(content_id); - takeover_release.notify_one(); - let projection = waiter.await.unwrap().unwrap(); - assert_eq!(projection.chunk_bytes(0).unwrap(), raw.as_slice()); - assert_flight_released(content_id); -} - -#[tokio::test] -async fn cancelled_waiter_does_not_cancel_or_leak_the_leader() { - let (content_id, raw) = unique_data(127); - let entered = Arc::new(Notify::new()); - let release = Arc::new(Notify::new()); - let leader = tokio::spawn({ - let entered = entered.clone(); - let release = release.clone(); - async move { - get_or_project(content_id, || async move { - entered.notify_one(); - release.notified().await; - Ok(raw.as_ref().clone()) - }) - .await - } - }); - started(&entered).await; - let waiter = tokio::spawn(async move { - get_or_project(content_id, || async { - panic!("cancelled waiter must not load while leader is active"); - }) - .await - }); - participants(content_id, 2).await; - waiter.abort(); - assert!(waiter.await.err().unwrap().is_cancelled()); - participants(content_id, 1).await; - release.notify_one(); - let projection = leader.await.unwrap().unwrap(); - assert_eq!(projection.map.file_content_id, content_id); - assert_flight_released(content_id); -} - -#[tokio::test] -async fn evicted_result_survives_leader_drop_until_joined_waiter_finishes() { - let (content_id, raw) = unique_data(131); - let calls = Arc::new(AtomicUsize::new(0)); - let entered = Arc::new(Notify::new()); - let release = Arc::new(Notify::new()); - let leader = tokio::spawn({ - let raw = raw.clone(); - let calls = calls.clone(); - let entered = entered.clone(); - let release = release.clone(); - async move { - get_or_project(content_id, || async move { - calls.fetch_add(1, Ordering::SeqCst); - entered.notify_one(); - release.notified().await; - Ok(raw.as_ref().clone()) - }) - .await - } - }); - started(&entered).await; - let registry = FLIGHTS.get().unwrap(); - let retained = FlightClaim::acquire(registry, content_id).unwrap(); - let result_weak; - { - // Queue a test-only lock before the real waiter, keeping its turn - // blocked until eviction and the leader's return owner are gone. - let queued = retained.flight.as_ref().unwrap().result.lock(); - tokio::pin!(queued); - assert!(futures::poll!(&mut queued).is_pending()); - let waiter = tokio::spawn({ - let raw = raw.clone(); - let calls = calls.clone(); - async move { - get_or_project(content_id, || async move { - calls.fetch_add(1, Ordering::SeqCst); - Ok(raw.as_ref().clone()) - }) - .await - } - }); - participants(content_id, 3).await; - release.notify_one(); - let projection = leader.await.unwrap().unwrap(); - let guard = queued.await; - result_weak = Arc::downgrade(&projection); - evict(content_id); - drop(projection); - assert_uncached(content_id); - assert!(result_weak.upgrade().is_some()); - drop(guard); - let shared = waiter.await.unwrap().unwrap(); - assert!(result_weak.ptr_eq(&Arc::downgrade(&shared))); - assert_eq!(shared.chunk_bytes(0).unwrap(), raw.as_slice()); - assert_eq!(calls.load(Ordering::SeqCst), 1); - drop(shared); - } - assert!(result_weak.upgrade().is_some()); - drop(retained); - assert!(result_weak.upgrade().is_none()); - assert_flight_released(content_id); - assert_uncached(content_id); - let reloaded = get_or_project(content_id, || async { - calls.fetch_add(1, Ordering::SeqCst); - Ok(raw.as_ref().clone()) - }) - .await - .unwrap(); - assert_eq!(calls.load(Ordering::SeqCst), 2); - assert_eq!(reloaded.chunk_bytes(0).unwrap(), raw.as_slice()); - assert_flight_released(content_id); -} - -#[test] -fn distinct_flights_are_bounded_but_existing_digest_can_join_at_capacity() { - let registry = Mutex::new(FlightRegistry::default()); - let mut claims = Vec::new(); - for index in 0..PROJECTION_FLIGHT_CAP { - claims.push(FlightClaim::acquire(®istry, [index as u8; 32]).unwrap()); - } - let joined = FlightClaim::acquire(®istry, [42; 32]).unwrap(); - assert!(Arc::ptr_eq( - claims[42].flight.as_ref().unwrap(), - joined.flight.as_ref().unwrap() - )); - let error = FlightClaim::acquire(®istry, [255; 32]).err().unwrap(); - assert_eq!(error.code, SnapshotErrorCode::LimitExceeded); - drop(claims); - assert_eq!(registry.lock().unwrap().entries.len(), 1); - drop(joined); - assert!(registry.lock().unwrap().entries.is_empty()); - let fresh = FlightClaim::acquire(®istry, [255; 32]).unwrap(); - drop(fresh); - assert!(registry.lock().unwrap().entries.is_empty()); -} - -#[test] -fn last_claim_removes_only_its_exact_registered_gate() { - let registry = Mutex::new(FlightRegistry::default()); - let content_id = [91; 32]; - let old = FlightClaim::acquire(®istry, content_id).unwrap(); - let replacement = Arc::new(ProjectionFlight { - result: tokio::sync::Mutex::new(None), - }); - registry - .lock() - .unwrap() - .entries - .insert(content_id, Arc::downgrade(&replacement)); - drop(old); - assert!(registry.lock().unwrap().entries[&content_id].ptr_eq(&Arc::downgrade(&replacement))); - let joined = FlightClaim::acquire(®istry, content_id).unwrap(); - assert!(Arc::ptr_eq(joined.flight.as_ref().unwrap(), &replacement)); - drop(replacement); - drop(joined); - assert!(registry.lock().unwrap().entries.is_empty()); - registry - .lock() - .unwrap() - .entries - .insert(content_id, Weak::new()); - let reclaimed = FlightClaim::acquire(®istry, content_id).unwrap(); - drop(reclaimed); - assert!(registry.lock().unwrap().entries.is_empty()); -} - -#[test] -fn concurrent_final_claim_drops_release_registry_capacity() { - let registry = Mutex::new(FlightRegistry::default()); - for index in 0..64 { - let content_id = [index; 32]; - let first = FlightClaim::acquire(®istry, content_id).unwrap(); - let second = FlightClaim::acquire(®istry, content_id).unwrap(); - let barrier = Barrier::new(2); - std::thread::scope(|scope| { - let barrier = &barrier; - scope.spawn(move || { - barrier.wait(); - drop(first); - }); - scope.spawn(move || { - barrier.wait(); - drop(second); - }); - }); - assert!(registry.lock().unwrap().entries.is_empty()); - } -} diff --git a/src/ceres/snapshot/chunks.rs b/src/ceres/snapshot/chunks.rs index 8999a45e..8346cec2 100644 --- a/src/ceres/snapshot/chunks.rs +++ b/src/ceres/snapshot/chunks.rs @@ -1,34 +1,132 @@ -//! On-the-fly, range-readable chunk projection for one file (spec 07). -//! -//! Persistent segment/locator storage is T04/T12 work; this identity slice -//! builds the MCM2 map and MCL2 leaves from a fixed view's verified blob and -//! retains inline bytes through 512 MiB and only verified map metadata for -//! larger files. Large CHUNK reads use the current request's strict raw range -//! source. This is a reproducible process cache, not a durable locator. -//! -//! The cache is an optimization, never the authority: every projected file -//! is re-hashed against the `content_id` the fixed view advertised, and a -//! cache miss simply rebuilds from Git. Entries are addressed by content -//! digest, never by request path. +//! Cold full-stream chunk-map verifier. Actual content callers use immutable +//! source receipts and selected persisted pages, never a digest-only cache. -use std::{ - collections::{HashMap, VecDeque}, - sync::{Arc, Mutex, OnceLock, Weak}, -}; +use std::sync::Arc; use mst2_codec::chunkmap::{CHUNK_SIZE, CHUNKS_PER_PAGE, ChunkLeaf, ChunkMap}; use sha2::{Digest, Sha256}; -use super::content_budget::{MemoryLease, projection_budget}; +use super::content_budget::MemoryLease; use crate::ceres::snapshot::error::{SnapshotError, SnapshotErrorCode}; #[path = "chunks_stream.rs"] mod streaming; -pub use streaming::get_or_project_stream; +/// Full-stream admission for one exact source, independent of the digest cache. +/// This opaque value is the only production input to durable map installation. +pub(crate) struct VerifiedSourceChunkMap { + source: ChunkMapSource, + projection: ChunkProjection, +} + +#[derive(Debug, Clone, PartialEq, Eq)] +pub(crate) struct ChunkMapSource { + fact: crate::callisto::mst2_verified_object::Model, +} + +impl ChunkMapSource { + pub(crate) fn from_fact( + fact: crate::callisto::mst2_verified_object::Model, + oid: &str, + ) -> Result { + if fact.id <= 0 + || fact.storage_domain != "git" + || fact.object_kind != "blob" + || fact.git_oid != oid + || ![40, 64].contains(&oid.len()) + || !oid + .bytes() + .all(|b| b.is_ascii_hexdigit() && !b.is_ascii_uppercase()) + || fact.state != "VERIFIED" + || fact.verification_version + != crate::jupiter::storage::mono_storage::MST2_VERIFICATION_VERSION + || fact.raw_sha256.len() != 32 + || fact.size <= 0 + || fact.size as u64 > 8 * 1024 * 1024 * 1024 * 1024 + { + return Err(SnapshotError::new( + SnapshotErrorCode::IntegrityError, + "invalid exact chunk map source fact", + )); + } + Ok(Self { fact }) + } + + pub(crate) fn fact(&self) -> &crate::callisto::mst2_verified_object::Model { + &self.fact + } + + pub(crate) fn canonical_bytes(&self) -> Result, SnapshotError> { + let f = &self.fact; + serde_json::to_vec(&( + f.id, + &f.storage_domain, + &f.git_oid, + &f.object_kind, + &f.raw_sha256, + f.size, + f.verification_version, + &f.state, + f.created_at, + )) + .map_err(|_| internal("chunk map source encoding failed")) + } +} + +impl VerifiedSourceChunkMap { + pub(crate) async fn verify( + handler: &T, + source: ChunkMapSource, + budget: &Arc, + ) -> Result { + let digest: [u8; 32] = source + .fact + .raw_sha256 + .as_slice() + .try_into() + .map_err(|_| internal("chunk map source digest shape changed"))?; + let projection = streaming::build_source_stream( + digest, + source.fact.size as u64, + || async { + handler + .get_raw_blob_stream_by_hash(&source.fact.git_oid) + .await + .map_err(|error| { + let code = match error { + crate::common::errors::MegaError::ObjStorageNotFound(_) => { + SnapshotErrorCode::ObjectUnavailable + } + crate::common::errors::MegaError::ObjStorageInconsistent(_) => { + SnapshotErrorCode::IntegrityError + } + _ => SnapshotErrorCode::Internal, + }; + SnapshotError::new(code, "exact chunk map source could not be read") + }) + }, + budget, + ) + .await?; + Ok(Self { source, projection }) + } + + pub(crate) fn source(&self) -> &ChunkMapSource { + &self.source + } + pub(crate) fn map(&self) -> &ChunkMap { + &self.projection.map + } + pub(crate) fn leaves(&self) -> &[ChunkLeaf] { + &self.projection.leaves + } + pub(crate) fn leaf_hashes(&self) -> &[[u8; 32]] { + &self.projection.leaf_hashes + } +} -pub(crate) fn projection_reservation_bytes(size: u64) -> Result { - streaming::reserved_bytes(size) +pub(crate) fn map_build_reservation_bytes(size: u64) -> Result { + streaming::reserved_map_bytes(size) } /// One file's verified range-readable projection. @@ -111,6 +209,7 @@ impl ChunkProjection { }) } + #[cfg(test)] pub fn has_inline_bytes(&self) -> bool { !self.raw.is_empty() } @@ -133,6 +232,7 @@ impl ChunkProjection { .sum::() } + #[cfg(test)] pub fn verify_chunk(&self, index: u64, bytes: &[u8]) -> Result<(), SnapshotError> { let want_len = self.map.chunk_len(index).map_err(codec_err)?; if bytes.len() as u64 != want_len { @@ -153,11 +253,13 @@ impl ChunkProjection { Ok(()) } + #[cfg(test)] pub fn page_count(&self) -> u64 { self.map.page_count } /// One MCL2 leaf plus its bottom-up proof toward `pages_root`. + #[cfg(test)] pub fn leaf_and_proof( &self, page_index: u64, @@ -179,6 +281,7 @@ impl ChunkProjection { /// Raw bytes of chunk `index`, length-checked against the map (spec 07 /// §4: full 1 MiB chunks except the positive-length remainder). + #[cfg(test)] pub fn chunk_bytes(&self, index: u64) -> Result<&[u8], SnapshotError> { let want_len = self.map.chunk_len(index).map_err(codec_err)?; let start = (index as usize) @@ -222,220 +325,6 @@ fn internal(m: &str) -> SnapshotError { SnapshotError::new(SnapshotErrorCode::Internal, m) } -/// Bounded process-wide cache of verified projections. Eviction is pure -/// memory reclaim; the projection is reproducible from Git, so an evicted -/// entry is rebuilt, never an error to the client. -static STAGED: OnceLock> = OnceLock::new(); - -/// 512 MiB cap for staged file bytes in this identity slice. The real -/// persistent RAW_OBJECT locator layout replaces this (spec 10 §4). -const STAGED_CAP_BYTES: usize = 512 * 1024 * 1024; -const STAGED_MAX_ENTRIES: usize = 4096; - -struct ProjectionCache { - entries: HashMap<[u8; 32], std::sync::Arc>, - /// Insertion/last-used order for FIFO reclaim. - order: VecDeque<[u8; 32]>, - total_bytes: usize, -} - -impl ProjectionCache { - fn new() -> Self { - ProjectionCache { - entries: HashMap::new(), - order: VecDeque::new(), - total_bytes: 0, - } - } - - fn get(&mut self, id: [u8; 32]) -> Option> { - self.entries.get(&id).cloned() - } - - fn evict_oldest(&mut self) -> Option> { - while let Some(victim) = self.order.pop_front() { - if let Some(projection) = self.entries.remove(&victim) { - self.total_bytes -= projection.retained_bytes(); - return Some(projection); - } - } - None - } - - fn put(&mut self, proj: std::sync::Arc) -> Vec> { - let mut evicted = Vec::new(); - let id = proj.map.file_content_id; - if self.entries.contains_key(&id) { - return evicted; - } - // Reclaim oldest entries until the new one fits. A file larger than - // the cap alone is not cached between flights but still - // served correctly. - let bytes = proj.retained_bytes(); - if bytes > STAGED_CAP_BYTES { - return evicted; - } - while (self.total_bytes + bytes > STAGED_CAP_BYTES - || self.entries.len() >= STAGED_MAX_ENTRIES) - && let Some(victim) = self.evict_oldest() - { - evicted.push(victim); - } - self.total_bytes += bytes; - self.order.push_back(id); - self.entries.insert(id, proj); - evicted - } -} - -static FLIGHTS: OnceLock> = OnceLock::new(); -const PROJECTION_FLIGHT_CAP: usize = 128; - -struct ProjectionFlight { - result: tokio::sync::Mutex>>, -} - -#[derive(Default)] -struct FlightRegistry { - entries: HashMap<[u8; 32], Weak>, -} - -struct FlightClaim<'a> { - content_id: [u8; 32], - flight: Option>, - registry: &'a Mutex, -} - -impl<'a> FlightClaim<'a> { - fn acquire( - registry: &'a Mutex, - content_id: [u8; 32], - ) -> Result { - let mut state = registry - .lock() - .map_err(|_| internal("chunk projection flight registry lock poisoned"))?; - if let Some(flight) = state.entries.get(&content_id).and_then(Weak::upgrade) { - return Ok(Self { - content_id, - flight: Some(flight), - registry, - }); - } - state.entries.remove(&content_id); - if state.entries.len() >= PROJECTION_FLIGHT_CAP { - return Err(SnapshotError::new( - SnapshotErrorCode::LimitExceeded, - "too many distinct chunk projections in flight", - )); - } - let flight = Arc::new(ProjectionFlight { - result: tokio::sync::Mutex::new(None), - }); - state.entries.insert(content_id, Arc::downgrade(&flight)); - Ok(Self { - content_id, - flight: Some(flight), - registry, - }) - } -} - -impl Drop for FlightClaim<'_> { - fn drop(&mut self) { - let Some(flight) = self.flight.take() else { - return; - }; - let Ok(mut state) = self.registry.lock() else { - return; - }; - if Arc::strong_count(&flight) == 1 { - if state - .entries - .get(&self.content_id) - .is_some_and(|registered| registered.ptr_eq(&Arc::downgrade(&flight))) - { - state.entries.remove(&self.content_id); - } - drop(state); - drop(flight); - } else { - // Serialize owner release with admission and other final drops. - // Another participant keeps the potentially large result alive. - drop(flight); - drop(state); - } - } -} - -/// Return the cached projection, or build one via `load` (which must resolve -/// the file in the fixed view and return its verified bytes). Concurrent -/// misses for the same digest share one successful load and projection. -#[cfg(test)] -pub async fn get_or_project( - content_id: [u8; 32], - load: F, -) -> Result, SnapshotError> -where - F: FnOnce() -> Fut, - Fut: std::future::Future, SnapshotError>>, -{ - get_or_project_with(content_id, || async { - ChunkProjection::build(content_id, load().await?) - }) - .await -} - -async fn get_or_project_with( - content_id: [u8; 32], - build: F, -) -> Result, SnapshotError> -where - F: FnOnce() -> Fut, - Fut: std::future::Future>, -{ - let cache = STAGED.get_or_init(|| Mutex::new(ProjectionCache::new())); - if let Some(p) = cache - .lock() - .map_err(|_| internal("chunk projection cache lock poisoned"))? - .get(content_id) - { - return Ok(p); - } - - let registry = FLIGHTS.get_or_init(|| Mutex::new(FlightRegistry::default())); - let claim = FlightClaim::acquire(registry, content_id)?; - let flight = claim - .flight - .as_ref() - .ok_or_else(|| internal("chunk projection flight claim released"))?; - let mut result = flight.result.lock().await; - if let Some(p) = cache - .lock() - .map_err(|_| internal("chunk projection cache lock poisoned"))? - .get(content_id) - { - *result = Some(p.clone()); - return Ok(p); - } - if let Some(p) = result.as_ref() { - return Ok(p.clone()); - } - let proj = Arc::new(build().await?); - let evicted = cache - .lock() - .map_err(|_| internal("chunk projection cache lock poisoned"))? - .put(proj.clone()); - // Final projection/credit drops can be expensive and must not hold the - // process cache mutex. Eviction does not refund other Arc owners. - drop(evicted); - *result = Some(proj.clone()); - Ok(proj) -} - -#[cfg(test)] -#[path = "chunk_singleflight_tests.rs"] -mod singleflight_tests; - #[cfg(test)] mod tests { use super::*; diff --git a/src/ceres/snapshot/chunks_stream.rs b/src/ceres/snapshot/chunks_stream.rs index 56f007fd..ab19e5b5 100644 --- a/src/ceres/snapshot/chunks_stream.rs +++ b/src/ceres/snapshot/chunks_stream.rs @@ -1,3 +1,5 @@ +use std::sync::{Arc, OnceLock}; + use futures::StreamExt; use tokio::sync::Semaphore; @@ -7,9 +9,20 @@ use crate::orbit_api::object_storage::ObjectByteStream; const STREAM_ITEM_MAX_BYTES: usize = 8 * 1024 * 1024; const MAX_FILE_BYTES: u64 = 8 * 1024 * 1024 * 1024 * 1024; const MAX_BUILDERS: usize = 4; +#[cfg(test)] +const STAGED_CAP_BYTES: usize = 512 * 1024 * 1024; const CONSTRUCTION_ALLOWANCE: usize = 64 * 1024; +#[cfg(test)] pub(super) fn reserved_bytes(size: u64) -> Result { + reserved_bytes_for(size, size <= STAGED_CAP_BYTES as u64) +} + +pub(super) fn reserved_map_bytes(size: u64) -> Result { + reserved_bytes_for(size, false) +} + +fn reserved_bytes_for(size: u64, inline: bool) -> Result { if size == 0 || size > MAX_FILE_BYTES { return Err(SnapshotError::new( if size == 0 { @@ -22,11 +35,7 @@ pub(super) fn reserved_bytes(size: u64) -> Result { } let chunks = size.div_ceil(CHUNK_SIZE as u64); let pages = chunks.div_ceil(CHUNKS_PER_PAGE as u64); - let inline = if size <= STAGED_CAP_BYTES as u64 { - size - } else { - 0 - }; + let inline = if inline { size } else { 0 }; let bytes = chunks .checked_mul(32) .and_then(|n| { @@ -41,65 +50,50 @@ pub(super) fn reserved_bytes(size: u64) -> Result { Ok(bytes) } -fn reserve_projection( +#[cfg(test)] +async fn build_stream( + content_id: [u8; 32], size: u64, - budget: &Arc, - cache: &Mutex, -) -> Result { - let bytes = reserved_bytes(size)?; - loop { - match budget.reserve(bytes) { - Ok(lease) => return Ok(lease), - Err(error) => { - if error.code != SnapshotErrorCode::TemporaryUnavailable { - return Err(error); - } - let victim = cache - .lock() - .map_err(|_| internal("chunk projection cache lock poisoned"))? - .evict_oldest(); - let Some(victim) = victim else { - return Err(error); - }; - drop(victim); - } - } - } + input: ObjectByteStream, + lease: MemoryLease, +) -> Result { + build_stream_with_inline( + content_id, + size, + input, + lease, + size <= STAGED_CAP_BYTES as u64, + ) + .await } -pub async fn get_or_project_stream( +pub(super) async fn build_source_stream( content_id: [u8; 32], size: u64, open: F, -) -> Result, SnapshotError> + budget: &Arc, +) -> Result where F: FnOnce() -> Fut, Fut: std::future::Future>, { - // Profile arithmetic also runs on hits; fixed facts remain per-request. - reserved_bytes(size)?; - get_or_project_with(content_id, || async { - static BUILDERS: OnceLock> = OnceLock::new(); - build_with_resources( - content_id, - size, - open, - projection_budget(), - BUILDERS.get_or_init(|| Arc::new(Semaphore::new(MAX_BUILDERS))), - STAGED.get_or_init(|| Mutex::new(ProjectionCache::new())), - ) - .await - }) + static BUILDERS: OnceLock> = OnceLock::new(); + build_source_with_resources( + content_id, + size, + open, + budget, + BUILDERS.get_or_init(|| Arc::new(Semaphore::new(MAX_BUILDERS))), + ) .await } -async fn build_with_resources( +async fn build_source_with_resources( content_id: [u8; 32], size: u64, open: F, budget: &Arc, builders: &Arc, - cache: &Mutex, ) -> Result where F: FnOnce() -> Fut, @@ -108,20 +102,39 @@ where let _builder = builders.clone().try_acquire_owned().map_err(|_| { SnapshotError::new( SnapshotErrorCode::TemporaryUnavailable, - "chunk projection builders are occupied", + "chunk map builders are occupied", ) })?; - let lease = reserve_projection(size, budget, cache)?; - // Source body I/O starts only after builder and retained-memory credit. + let lease = budget.reserve(source_reservation_bytes(size)?)?; let input = open().await?; - build_stream(content_id, size, input, lease).await + build_stream_with_inline(content_id, size, input, lease, false).await } -async fn build_stream( +pub(super) fn source_reservation_bytes(size: u64) -> Result { + let map_bytes = reserved_map_bytes(size)?; + let pages = size + .div_ceil(CHUNK_SIZE as u64) + .div_ceil(CHUNKS_PER_PAGE as u64); + let nodes = pages + .checked_mul(2) + .and_then(|n| n.checked_sub(1)) + .and_then(|n| { + n.checked_mul(std::mem::size_of::() as u64) + }) + .and_then(|n| usize::try_from(n).ok()) + .ok_or_else(|| internal("chunk map install reservation overflow"))?; + map_bytes + .checked_add(nodes) + .and_then(|n| n.checked_add(4 * 1024 * 1024)) + .ok_or_else(|| internal("chunk map install reservation overflow")) +} + +async fn build_stream_with_inline( content_id: [u8; 32], size: u64, mut input: ObjectByteStream, lease: MemoryLease, + inline: bool, ) -> Result { let chunk_count = size.div_ceil(CHUNK_SIZE as u64); let page_count = chunk_count.div_ceil(CHUNKS_PER_PAGE as u64); @@ -129,7 +142,6 @@ async fn build_stream( let mut leaves = Vec::new(); let mut leaf_hashes = Vec::new(); let mut current = Vec::new(); - let inline = size <= STAGED_CAP_BYTES as u64; if inline { raw.try_reserve_exact(size as usize) .map_err(allocation_error)?; diff --git a/src/ceres/snapshot/chunks_stream_tests.rs b/src/ceres/snapshot/chunks_stream_tests.rs index 2ccebae8..2cf7fb53 100644 --- a/src/ceres/snapshot/chunks_stream_tests.rs +++ b/src/ceres/snapshot/chunks_stream_tests.rs @@ -14,6 +14,112 @@ fn stream(parts: Vec>) -> ObjectByteStream { Box::pin(futures::stream::iter(parts)) } +#[tokio::test] +async fn exact_source_builder_admits_all_install_workspace_before_open_and_cancellation_releases_it() + { + let raw = Bytes::from_static(b"exact source body"); + let digest: [u8; 32] = Sha256::digest(&raw).into(); + let weight = source_reservation_bytes(raw.len() as u64).unwrap(); + assert!(weight > 4 * 1024 * 1024); + let budget = MemoryBudget::new(weight); + let builders = Arc::new(Semaphore::new(1)); + let opens = Arc::new(AtomicUsize::new(0)); + let held = budget.reserve(weight).unwrap(); + let error = build_source_with_resources( + digest, + raw.len() as u64, + || async { + opens.fetch_add(1, Ordering::SeqCst); + Ok(stream(vec![Ok(raw.clone())])) + }, + &budget, + &builders, + ) + .await + .err() + .unwrap(); + assert_eq!(error.code, SnapshotErrorCode::TemporaryUnavailable); + assert_eq!(opens.load(Ordering::SeqCst), 0); + assert_eq!(builders.available_permits(), 1); + drop(held); + let held_builder = builders.clone().acquire_owned().await.unwrap(); + assert_eq!( + build_source_with_resources( + digest, + raw.len() as u64, + || async { + opens.fetch_add(1, Ordering::SeqCst); + Ok(stream(vec![Ok(raw.clone())])) + }, + &budget, + &builders + ) + .await + .err() + .unwrap() + .code, + SnapshotErrorCode::TemporaryUnavailable + ); + assert_eq!(opens.load(Ordering::SeqCst), 0); + assert_eq!(budget.used(), 0); + drop(held_builder); + let entered = Arc::new(Notify::new()); + let task_budget = budget.clone(); + let task_builders = builders.clone(); + let task_entered = entered.clone(); + let task_opens = opens.clone(); + let drops = Arc::new(AtomicUsize::new(0)); + let producer_owner = DropCount(drops.clone()); + let size = raw.len() as u64; + let task = tokio::spawn(async move { + build_source_with_resources( + digest, + size, + || async { + task_opens.fetch_add(1, Ordering::SeqCst); + task_entered.notify_one(); + Ok(Box::pin(futures::stream::unfold( + producer_owner, + |owner| async move { + futures::future::pending::<()>().await; + Some((Ok(Bytes::new()), owner)) + }, + )) as ObjectByteStream) + }, + &task_budget, + &task_builders, + ) + .await + }); + timeout(Duration::from_secs(5), entered.notified()) + .await + .unwrap(); + assert_eq!(budget.used(), weight); + assert_eq!(builders.available_permits(), 0); + task.abort(); + assert!(task.await.err().unwrap().is_cancelled()); + assert_eq!(budget.used(), 0); + assert_eq!(builders.available_permits(), 1); + assert_eq!(drops.load(Ordering::SeqCst), 1); + let projection = build_source_with_resources( + digest, + size, + || async { Ok(stream(vec![Ok(raw.clone())])) }, + &budget, + &builders, + ) + .await + .unwrap(); + assert!(!projection.has_inline_bytes()); + assert_eq!(projection.map.file_content_id, digest); + assert_eq!(opens.load(Ordering::SeqCst), 2); + assert_eq!(builders.available_permits(), 1); + assert_eq!(budget.used(), weight); + projection.verify_chunk(0, &raw).unwrap(); + drop(projection); + assert_eq!(budget.used(), 0); +} + async fn project(raw: &[u8], parts: Vec>) -> ChunkProjection { let budget = MemoryBudget::new(reserved_bytes(raw.len() as u64).unwrap()); let lease = budget @@ -205,36 +311,6 @@ async fn visible_producer_item_limit_does_not_collect_a_large_item() { assert_eq!(budget.used(), 0); } -#[tokio::test] -async fn evicted_projection_remains_charged_until_last_reader_drops() { - let raw = b"owned credits"; - let budget = MemoryBudget::new(reserved_bytes(raw.len() as u64).unwrap()); - let lease = budget - .reserve(reserved_bytes(raw.len() as u64).unwrap()) - .unwrap(); - let projection = Arc::new( - build_stream( - Sha256::digest(raw).into(), - raw.len() as u64, - stream(vec![Ok(Bytes::from_static(raw))]), - lease, - ) - .await - .unwrap(), - ); - let weight = projection.retained_bytes(); - let mut cache = ProjectionCache::new(); - assert!(cache.put(projection.clone()).is_empty()); - let reader = projection.clone(); - drop(projection); - drop(cache.evict_oldest()); - assert_eq!(cache.total_bytes, 0); - assert_eq!(budget.used(), weight); - assert!(budget.reserve(1).is_err()); - drop(reader); - assert_eq!(budget.used(), 0); -} - struct DropCount(Arc); impl Drop for DropCount { fn drop(&mut self) { @@ -242,139 +318,6 @@ impl Drop for DropCount { } } -#[tokio::test] -async fn production_admission_rejects_live_credit_and_builder_overload_before_open() { - let weight = reserved_bytes(3).unwrap(); - let cache = Mutex::new(ProjectionCache::new()); - let budget = MemoryBudget::new(weight); - let builders = Arc::new(Semaphore::new(MAX_BUILDERS)); - let calls = AtomicUsize::new(0); - let held = budget.reserve(weight).unwrap(); - let error = build_with_resources( - Sha256::digest(b"abc").into(), - 3, - || async { - calls.fetch_add(1, Ordering::SeqCst); - Ok(stream(vec![Ok(Bytes::from_static(b"abc"))])) - }, - &budget, - &builders, - &cache, - ) - .await - .err() - .unwrap(); - assert_eq!(error.code, SnapshotErrorCode::TemporaryUnavailable); - assert_eq!(calls.load(Ordering::SeqCst), 0); - assert_eq!(builders.available_permits(), MAX_BUILDERS); - drop(held); - let mut workers = Vec::new(); - for _ in 0..MAX_BUILDERS { - workers.push(builders.clone().try_acquire_owned().unwrap()); - } - let error = build_with_resources( - Sha256::digest(b"abc").into(), - 3, - || async { - calls.fetch_add(1, Ordering::SeqCst); - Ok(stream(vec![Ok(Bytes::from_static(b"abc"))])) - }, - &budget, - &builders, - &cache, - ) - .await - .err() - .unwrap(); - assert_eq!(error.code, SnapshotErrorCode::TemporaryUnavailable); - assert_eq!(calls.load(Ordering::SeqCst), 0); - assert_eq!(budget.used(), 0); - drop(workers); - let projection = build_with_resources( - Sha256::digest(b"abc").into(), - 3, - || async { - calls.fetch_add(1, Ordering::SeqCst); - Ok(stream(vec![Ok(Bytes::from_static(b"abc"))])) - }, - &budget, - &builders, - &cache, - ) - .await - .unwrap(); - assert_eq!(calls.load(Ordering::SeqCst), 1); - assert_eq!(budget.used(), weight); - assert_eq!(builders.available_permits(), MAX_BUILDERS); - drop(projection); - assert_eq!(budget.used(), 0); -} - -#[tokio::test] -async fn cancelled_stream_owner_refunds_after_stream_drop_and_same_flight_waiter_takes_over() { - let raw = Bytes::from(uuid::Uuid::new_v4().as_bytes().to_vec()); - let id: [u8; 32] = Sha256::digest(&raw).into(); - let entered = Arc::new(Notify::new()); - let drops = Arc::new(AtomicUsize::new(0)); - let leader = tokio::spawn({ - let entered = entered.clone(); - let drops = drops.clone(); - async move { - get_or_project_stream(id, 16, || async { - let input: ObjectByteStream = Box::pin(futures::stream::unfold( - (entered, DropCount(drops)), - |(entered, owner)| async move { - entered.notify_one(); - std::future::pending::<()>().await; - Some((Ok(Bytes::new()), (entered, owner))) - }, - )); - Ok(input) - }) - .await - } - }); - timeout(Duration::from_secs(5), entered.notified()) - .await - .unwrap(); - let waiter = tokio::spawn(async move { - get_or_project_stream(id, 16, || async { - let input: ObjectByteStream = stream(vec![Ok(raw)]); - Ok(input) - }) - .await - }); - timeout(Duration::from_secs(5), async { - loop { - if FLIGHTS - .get() - .unwrap() - .lock() - .unwrap() - .entries - .get(&id) - .map_or(0, Weak::strong_count) - == 2 - { - break; - } - tokio::task::yield_now().await; - } - }) - .await - .unwrap(); - leader.abort(); - assert!(leader.await.err().unwrap().is_cancelled()); - assert_eq!(drops.load(Ordering::SeqCst), 1); - let projection = timeout(Duration::from_secs(5), waiter) - .await - .unwrap() - .unwrap() - .unwrap(); - assert_eq!(projection.map.file_content_id, id); - assert_eq!(projection.chunk_bytes(0).unwrap().len(), 16); -} - #[test] fn maximum_file_metadata_fits_and_protocol_or_quota_rejection_needs_no_source() { assert!(reserved_bytes(MAX_FILE_BYTES).unwrap() < 300 * 1024 * 1024); diff --git a/src/ceres/snapshot/mod.rs b/src/ceres/snapshot/mod.rs index 532d1da6..3fae7f32 100644 --- a/src/ceres/snapshot/mod.rs +++ b/src/ceres/snapshot/mod.rs @@ -3,6 +3,8 @@ //! Serving remains native-only. The independent namespace index is an unwired //! composition seam; identity, attestation and publication integration remain //! separate gates. Fixed-source readers must never look up current refs. +pub(crate) mod chunk_map_gate; +pub(crate) mod chunk_map_index; pub mod chunks; pub(crate) mod content_budget; pub mod descriptor; diff --git a/src/jupiter/migration/m20261008_000100_add_mst2_chunk_maps.rs b/src/jupiter/migration/m20261008_000100_add_mst2_chunk_maps.rs new file mode 100644 index 00000000..cd597a16 --- /dev/null +++ b/src/jupiter/migration/m20261008_000100_add_mst2_chunk_maps.rs @@ -0,0 +1,28 @@ +//! Append-only, independently authenticated chunk-map read indexes. + +use sea_orm::{ConnectionTrait, DbBackend}; +use sea_orm_migration::prelude::*; + +#[derive(DeriveMigrationName)] +pub struct Migration; + +#[async_trait::async_trait] +impl MigrationTrait for Migration { + async fn up(&self, manager: &SchemaManager) -> Result<(), DbErr> { + if manager.get_database_backend() != DbBackend::Postgres { + return Err(DbErr::Custom( + "persisted chunk maps require primary PostgreSQL".into(), + )); + } + manager + .get_connection() + .execute_unprepared(include_str!("m20261008_000100_chunk_maps.sql")) + .await?; + Ok(()) + } + + async fn down(&self, _manager: &SchemaManager) -> Result<(), DbErr> { + // No collector or migration rollback may silently discard receipts. + Ok(()) + } +} diff --git a/src/jupiter/migration/m20261008_000100_chunk_maps.sql b/src/jupiter/migration/m20261008_000100_chunk_maps.sql new file mode 100644 index 00000000..75b452d0 --- /dev/null +++ b/src/jupiter/migration/m20261008_000100_chunk_maps.sql @@ -0,0 +1,90 @@ +CREATE TABLE mst2_chunk_map ( + map_id bytea PRIMARY KEY CHECK (pg_catalog.octet_length(map_id)=32), + descriptor bytea NOT NULL CHECK (pg_catalog.octet_length(descriptor)=100), + page_count integer NOT NULL CHECK (page_count BETWEEN 1 AND 32768), + pages_root bytea NOT NULL CHECK (pg_catalog.octet_length(pages_root)=32) +); +CREATE TABLE mst2_chunk_map_leaf ( + map_id bytea NOT NULL REFERENCES mst2_chunk_map(map_id), + page_index integer NOT NULL CHECK (page_index BETWEEN 0 AND 32767), + payload bytea NOT NULL CHECK (pg_catalog.octet_length(payload) BETWEEN 48 AND 8208), + PRIMARY KEY(map_id,page_index) +); +CREATE TABLE mst2_chunk_map_node ( + map_id bytea NOT NULL REFERENCES mst2_chunk_map(map_id), + first_page integer NOT NULL CHECK (first_page BETWEEN 0 AND 32767), + page_count integer NOT NULL CHECK (page_count BETWEEN 1 AND 32768), + digest bytea NOT NULL CHECK (pg_catalog.octet_length(digest)=32), + PRIMARY KEY(map_id,first_page,page_count), + CHECK (first_page+page_count<=32768) +); +CREATE TABLE mst2_chunk_map_source ( + storage_domain text NOT NULL CHECK (storage_domain='git'), + git_oid text NOT NULL CHECK (git_oid ~ '^([0-9a-f]{40}|[0-9a-f]{64})$'), + object_kind text NOT NULL CHECK (object_kind='blob'), + fact_id bigint NOT NULL CHECK (fact_id>0), + source_id bytea NOT NULL UNIQUE CHECK (pg_catalog.octet_length(source_id)=32), + source_bytes bytea NOT NULL CHECK (pg_catalog.octet_length(source_bytes) BETWEEN 1 AND 2048), + primary_scope bytea NOT NULL CHECK (pg_catalog.octet_length(primary_scope) BETWEEN 1 AND 1024), + map_id bytea NOT NULL REFERENCES mst2_chunk_map(map_id), + receipt_digest bytea NOT NULL CHECK (pg_catalog.octet_length(receipt_digest)=32), + PRIMARY KEY(storage_domain,git_oid,object_kind) +); + +-- These rows are indexes, not proof that any source body was consumed. The +-- trusted object writer alone publishes the independent immutable receipt. +CREATE FUNCTION mst2_chunk_map_immutable() RETURNS trigger LANGUAGE plpgsql AS $$ +BEGIN RAISE EXCEPTION 'chunk map indexes and source receipts are append-only'; END $$; + +CREATE FUNCTION mst2_chunk_map_primary() RETURNS trigger LANGUAGE plpgsql AS $$ +BEGIN + IF pg_catalog.current_schema() IS DISTINCT FROM TG_TABLE_SCHEMA + OR pg_catalog.pg_is_in_recovery() + OR pg_catalog.current_setting('transaction_isolation')<>'read committed' THEN + RAISE EXCEPTION 'chunk map insertion requires its actual primary schema and READ COMMITTED'; + END IF; + RETURN NULL; +END $$; + +CREATE FUNCTION mst2_chunk_map_complete() RETURNS trigger LANGUAGE plpgsql AS $$ +DECLARE m record; leaves bigint; nodes bigint; root bytea; first_leaf integer; last_leaf integer; +BEGIN + EXECUTE pg_catalog.format('SELECT * FROM %I.mst2_chunk_map WHERE map_id=$1',TG_TABLE_SCHEMA) + INTO m USING NEW.map_id; + EXECUTE pg_catalog.format('SELECT pg_catalog.count(*),pg_catalog.min(page_index),pg_catalog.max(page_index) FROM %I.mst2_chunk_map_leaf WHERE map_id=$1',TG_TABLE_SCHEMA) + INTO leaves,first_leaf,last_leaf USING NEW.map_id; + EXECUTE pg_catalog.format('SELECT pg_catalog.count(*) FROM %I.mst2_chunk_map_node WHERE map_id=$1',TG_TABLE_SCHEMA) + INTO nodes USING NEW.map_id; + EXECUTE pg_catalog.format('SELECT digest FROM %I.mst2_chunk_map_node WHERE map_id=$1 AND first_page=0 AND page_count=$2',TG_TABLE_SCHEMA) + INTO root USING NEW.map_id,m.page_count; + IF m.map_id IS NULL OR leaves<>m.page_count OR first_leaf<>0 OR last_leaf<>m.page_count-1 + OR nodes<>2*m.page_count-1 OR root IS DISTINCT FROM m.pages_root + OR pg_catalog.substring(m.descriptor,69,32) IS DISTINCT FROM m.pages_root + OR pg_catalog.sha256(pg_catalog.convert_to('mega.mst2.chunkmap','UTF8')||pg_catalog.decode('00','hex')||m.descriptor) IS DISTINCT FROM m.map_id THEN + RAISE EXCEPTION 'chunk map installation is incomplete or inconsistent'; + END IF; + RETURN NULL; +END $$; + +CREATE FUNCTION mst2_chunk_map_index_insert() RETURNS trigger LANGUAGE plpgsql AS $$ +DECLARE admitted boolean; +BEGIN + EXECUTE pg_catalog.format('SELECT EXISTS(SELECT 1 FROM %I.mst2_chunk_map_source WHERE map_id=$1)',TG_TABLE_SCHEMA) + INTO admitted USING NEW.map_id; + IF admitted THEN RAISE EXCEPTION 'admitted chunk map index cannot acquire more rows'; END IF; + RETURN NEW; +END $$; + +DO $$ DECLARE t text; BEGIN + FOREACH t IN ARRAY ARRAY['mst2_chunk_map','mst2_chunk_map_leaf','mst2_chunk_map_node','mst2_chunk_map_source'] LOOP + EXECUTE pg_catalog.format('CREATE TRIGGER mst2_chunk_map_primary BEFORE INSERT ON %I FOR EACH STATEMENT EXECUTE FUNCTION mst2_chunk_map_primary()',t); + EXECUTE pg_catalog.format('CREATE TRIGGER mst2_chunk_map_immutable BEFORE UPDATE OR DELETE ON %I FOR EACH ROW EXECUTE FUNCTION mst2_chunk_map_immutable()',t); + EXECUTE pg_catalog.format('CREATE TRIGGER mst2_chunk_map_no_truncate BEFORE TRUNCATE ON %I FOR EACH STATEMENT EXECUTE FUNCTION mst2_chunk_map_immutable()',t); + END LOOP; +END $$; +CREATE CONSTRAINT TRIGGER mst2_chunk_map_complete AFTER INSERT ON mst2_chunk_map_source + DEFERRABLE INITIALLY DEFERRED FOR EACH ROW EXECUTE FUNCTION mst2_chunk_map_complete(); +CREATE TRIGGER mst2_chunk_map_index_insert BEFORE INSERT ON mst2_chunk_map_leaf + FOR EACH ROW EXECUTE FUNCTION mst2_chunk_map_index_insert(); +CREATE TRIGGER mst2_chunk_map_index_insert BEFORE INSERT ON mst2_chunk_map_node + FOR EACH ROW EXECUTE FUNCTION mst2_chunk_map_index_insert(); diff --git a/src/jupiter/migration/mod.rs b/src/jupiter/migration/mod.rs index 2d667a39..f5981f02 100644 --- a/src/jupiter/migration/mod.rs +++ b/src/jupiter/migration/mod.rs @@ -151,6 +151,7 @@ mod m20261007_000300_add_mst2_metadata_lifetime_history; mod m20261007_000400_add_mst2_qualified_metadata_gc; mod m20261007_000500_add_mst2_install_capability; mod m20261007_000600_add_mst2_storage_routes; +mod m20261008_000100_add_mst2_chunk_maps; mod runner; pub use m20260905_000100_add_push_queue::ensure_queue_control_seed; pub use runner::apply_migrations; @@ -290,6 +291,7 @@ impl MigratorTrait for Migrator { Box::new(m20261007_000400_add_mst2_qualified_metadata_gc::Migration), Box::new(m20261007_000500_add_mst2_install_capability::Migration), Box::new(m20261007_000600_add_mst2_storage_routes::Migration), + Box::new(m20261008_000100_add_mst2_chunk_maps::Migration), ] } } diff --git a/src/jupiter/storage/mod.rs b/src/jupiter/storage/mod.rs index b130aa5f..2ea787de 100644 --- a/src/jupiter/storage/mod.rs +++ b/src/jupiter/storage/mod.rs @@ -18,6 +18,7 @@ pub mod media_paging_storage; pub mod mono_storage; pub(crate) mod mst2_publication_storage; pub mod mst2_retention; +pub(crate) mod native_chunk_map; pub mod native_metadata_install; pub(crate) mod native_publication_storage; pub(crate) mod native_snapshot_session; @@ -159,6 +160,8 @@ pub struct Storage { /// Derived native projection memoization, scoped to this storage assembly. /// Clones share it; independent databases/backends never share entries. pub(crate) native_projection_cache: Arc, + pub(crate) native_chunk_maps: + Arc>, pub(crate) native_snapshot_sessions: Arc>, pub(crate) projection_observation_sink: @@ -326,6 +329,7 @@ impl Storage { app_service: app_service.into(), native_projection_cache: Arc::default(), native_snapshot_sessions: Arc::default(), + native_chunk_maps: Arc::default(), projection_observation_sink: None, config_handle, config, @@ -704,6 +708,7 @@ impl Storage { app_service, native_projection_cache: Arc::default(), native_snapshot_sessions: Arc::default(), + native_chunk_maps: Arc::default(), projection_observation_sink: None, // app_service: AppService::mock(), cl_service: CLService::mock(), diff --git a/src/jupiter/storage/mono_storage.rs b/src/jupiter/storage/mono_storage.rs index 335869ac..5228c6fc 100644 --- a/src/jupiter/storage/mono_storage.rs +++ b/src/jupiter/storage/mono_storage.rs @@ -1789,12 +1789,43 @@ impl MonoStorage { if oids.is_empty() { return Ok(HashMap::new()); } - let rows = mst2_verified_object::Entity::find() - .filter(mst2_verified_object::Column::StorageDomain.eq("git")) - .filter(mst2_verified_object::Column::ObjectKind.eq("blob")) - .filter(mst2_verified_object::Column::GitOid.is_in(oids)) - .all(self.get_connection()) - .await?; + let connection = self.get_connection(); + let rows = if connection.get_database_backend() == sea_orm::DbBackend::Postgres { + use sea_orm::{DbBackend, FromQueryResult, Statement}; + let scope = connection.query_one_raw(Statement::from_string(DbBackend::Postgres, + "SELECT n.nspname AS schema FROM pg_catalog.pg_namespace n JOIN pg_catalog.pg_class c ON c.relnamespace=n.oid AND c.relname='mst2_verified_object' AND c.relkind='r' WHERE n.nspname=pg_catalog.current_schema() AND n.nspname NOT LIKE 'pg_temp_%'")) + .await?.ok_or_else(|| MegaError::Other("actual verified blob relation is missing".into()))?; + let schema: String = scope.try_get("", "schema")?; + let relation = format!("\"{}\".mst2_verified_object", schema.replace('"', "\"\"")); + let oids = serde_json::to_string(&oids) + .map_err(|error| MegaError::Other(error.to_string()))?; + let sql = format!( + "SELECT v.id,v.storage_domain,CASE WHEN pg_catalog.octet_length(v.git_oid) IN (40,64) THEN v.git_oid ELSE NULL END AS git_oid,v.object_kind,CASE WHEN pg_catalog.octet_length(v.raw_sha256)=32 THEN v.raw_sha256 ELSE NULL END AS raw_sha256,CASE WHEN v.size BETWEEN 0 AND {MST2_MAX_FILE_SIZE} THEN v.size ELSE NULL END AS size,CASE WHEN v.verification_version IN (1,{MST2_VERIFICATION_VERSION}) THEN v.verification_version ELSE NULL END AS verification_version,CASE WHEN v.state='VERIFIED' THEN v.state ELSE NULL END AS state,v.created_at FROM {relation} v WHERE v.storage_domain='git' AND v.object_kind='blob' AND v.git_oid IN (SELECT value FROM pg_catalog.jsonb_array_elements_text($1::jsonb))" + ); + connection + .query_all_raw(Statement::from_sql_and_values( + DbBackend::Postgres, + sql, + [oids.into()], + )) + .await? + .iter() + .map(|row| { + mst2_verified_object::Model::from_query_result(row, "").map_err(|_| { + MegaError::ObjStorageInconsistent( + "invalid bounded MST/2 verified blob record".into(), + ) + }) + }) + .collect::, _>>()? + } else { + mst2_verified_object::Entity::find() + .filter(mst2_verified_object::Column::StorageDomain.eq("git")) + .filter(mst2_verified_object::Column::ObjectKind.eq("blob")) + .filter(mst2_verified_object::Column::GitOid.is_in(oids)) + .all(connection) + .await? + }; for row in &rows { if row.state != "VERIFIED" || ![1, MST2_VERIFICATION_VERSION].contains(&row.verification_version) diff --git a/src/jupiter/storage/native_chunk_map.rs b/src/jupiter/storage/native_chunk_map.rs new file mode 100644 index 00000000..7dcbaa22 --- /dev/null +++ b/src/jupiter/storage/native_chunk_map.rs @@ -0,0 +1,684 @@ +//! Source-bound chunk-map receipts with indexed, authenticated selected pages. +//! +//! The object writer is trusted like the verified-object writer. A database +//! row, its checksum or a map's presence is never a full-source proof. Only +//! the opaque full-stream verifier can install a source receipt. This initial +//! store is append-only; bounded retention and collection remain separate work. + +use std::sync::Arc; + +use bytes::Bytes; +use futures::StreamExt; +use mst2_codec::chunkmap::{CHUNKS_PER_PAGE, ChunkLeaf, ChunkMap, ProofStep, verify_leaf}; +use sea_orm::{ + ConnectionTrait, DatabaseConnection, DatabaseTransaction, DbBackend, FromQueryResult, + IsolationLevel, QueryResult, Statement, TransactionTrait, +}; +use serde_json::json; +use sha2::{Digest, Sha256}; + +use super::object_storage::MegaObjectStorageWrapper; +use crate::{ + callisto::mst2_verified_object, + ceres::snapshot::{ + chunk_map_index::{ChunkMapNode, indexed_nodes, proof_intervals, selected_proof}, + chunks::{ChunkMapSource, VerifiedSourceChunkMap}, + content_budget::{MemoryBudget, MemoryLease, projection_budget}, + error::{SnapshotError, SnapshotErrorCode}, + }, + orbit_api::object_storage::{ObjectKey, ObjectMeta, ObjectNamespace}, +}; + +const DESCRIPTOR_CREDIT: usize = 32 * 1024; +const PAGE_CREDIT: usize = 64 * 1024; +const RECEIPT_DOMAIN: &[u8] = b"MST2-CHUNK-MAP-RECEIPT\0"; + +pub(crate) struct PostgresChunkMapRepository { + connection: DatabaseConnection, + primary_scope: Vec, + budget: Arc, + schema: String, +} + +pub(crate) struct PersistedChunkMap { + pub map: ChunkMap, + pub map_id: [u8; 32], + source: ChunkMapSource, + source_id: [u8; 32], + _memory: MemoryLease, +} + +impl PersistedChunkMap { + pub(crate) fn source_id(&self) -> [u8; 32] { + self.source_id + } +} + +pub(crate) struct AuthenticatedChunkPage { + pub leaf: ChunkLeaf, + pub proof: Vec, + _memory: MemoryLease, +} + +impl AuthenticatedChunkPage { + pub(crate) fn verify_chunk( + &self, + map: &ChunkMap, + index: u64, + bytes: &[u8], + ) -> Result<(), SnapshotError> { + let length = map + .chunk_len(index) + .map_err(|_| integrity("invalid requested chunk index"))?; + if index / CHUNKS_PER_PAGE as u64 != self.leaf.page_index || bytes.len() as u64 != length { + return Err(integrity( + "exact range length or selected chunk page disagrees", + )); + } + let slot = (index % CHUNKS_PER_PAGE as u64) as usize; + let expected = self + .leaf + .chunk_sha256 + .get(slot) + .ok_or_else(|| integrity("selected chunk digest is missing"))?; + let digest: [u8; 32] = Sha256::digest(bytes).into(); + if &digest != expected { + return Err(integrity( + "exact range digest disagrees with the authenticated selected page", + )); + } + Ok(()) + } +} + +impl PostgresChunkMapRepository { + pub(crate) async fn new(connection: DatabaseConnection) -> Result { + let schema = capture_schema(&connection).await?; + let primary_scope = read_scope(&connection, &schema).await?; + Ok(Self { + connection, + primary_scope, + budget: projection_budget().clone(), + schema, + }) + } + + pub(crate) async fn read( + &self, + source: &ChunkMapSource, + objects: &MegaObjectStorageWrapper, + ) -> Result>, SnapshotError> { + let memory = self.budget.reserve(DESCRIPTOR_CREDIT)?; + let txn = self.transaction().await?; + let result = async { + self.require_scope(&txn).await?; + require_current_source(&txn, source, &self.schema).await?; + let Some(row) = source_row(&txn, source, &self.schema).await? else { + return Ok(None); + }; + let (map, source_id) = self.validate_source_row(source, &row)?; + let bytes = receipt(&self.primary_scope, source, &map)?; + read_receipt(objects, &receipt_key(source_id), &bytes).await?; + Ok(Some(Arc::new(PersistedChunkMap { + map_id: map.map_id(), + map, + source: source.clone(), + source_id, + _memory: memory, + }))) + } + .await; + finish(txn, result).await + } + + /// Publication accepts no caller-provided map or "verified" flag. + pub(crate) async fn install( + &self, + verified: VerifiedSourceChunkMap, + objects: &MegaObjectStorageWrapper, + ) -> Result<(), SnapshotError> { + let source = verified.source(); + let map = verified.map(); + // The verifier reserved index and bounded SQL parameter workspace + // before opening the body. It owns those credits through publication. + let nodes = indexed_nodes(verified.leaf_hashes())?; + if nodes.last().map(|n| n.digest) != Some(map.pages_root) { + return Err(integrity( + "verified chunk map index disagrees with canonical root", + )); + } + let source_bytes = source.canonical_bytes()?; + let source_id = source_id(&self.primary_scope, &source_bytes); + let bytes = receipt(&self.primary_scope, source, map)?; + let key = receipt_key(source_id); + let txn = self.transaction().await?; + let result = async { + self.require_scope(&txn).await?; + require_current_source(&txn, source, &self.schema).await?; + if let Some(row) = source_row(&txn, source, &self.schema).await? { + let (stored, _) = self.validate_source_row(source, &row)?; + if stored != *map { return Err(integrity("immutable source map conflicts with reverified content")); } + read_receipt(objects, &key, &bytes).await?; + compare_complete_map(&txn, &verified, &nodes, &self.schema).await?; + return Ok(()); + } + // The complete source pass happened before this atomic create. + // A lost DB commit leaves an orphan trusted receipt, never an + // admitted partial map; exact re-verification can replay it. + objects.inner.put_metadata_atomic_create(&key, Bytes::copy_from_slice(&bytes), ObjectMeta { + size: bytes.len() as i64, ..Default::default() + }).await.map_err(storage_error)?; + read_receipt(objects, &key, &bytes).await?; + txn.execute_raw(stmt(&self.schema, "INSERT INTO mst2_chunk_map(map_id,descriptor,page_count,pages_root) VALUES($1,$2,$3,$4) ON CONFLICT(map_id) DO NOTHING", + [map.map_id().to_vec().into(), map.encode().into(), (map.page_count as i32).into(), map.pages_root.to_vec().into()])).await.map_err(db_error)?; + // Existing indexes must be compared, never repaired or overwritten. + let present = txn.query_one_raw(stmt(&self.schema, "SELECT EXISTS(SELECT 1 FROM mst2_chunk_map_source WHERE map_id=$1) AS sealed", [map.map_id().to_vec().into()])) + .await.map_err(db_error)?.ok_or_else(|| integrity("chunk map seal observation is missing"))? + .try_get::("", "sealed").map_err(db_error)?; + if !present { + for batch in verified.leaves().chunks(64) { + let encoded = encode_leaves(batch)?; + txn.execute_raw(stmt(&self.schema, "INSERT INTO mst2_chunk_map_leaf(map_id,page_index,payload) SELECT $1,p.page_index,pg_catalog.decode(p.payload,'hex') FROM pg_catalog.jsonb_to_recordset($2::jsonb) AS p(page_index integer,payload text) ON CONFLICT(map_id,page_index) DO NOTHING", + [map.map_id().to_vec().into(), encoded.into()])).await.map_err(db_error)?; + } + for batch in nodes.chunks(1024) { + let encoded = encode_nodes(batch)?; + txn.execute_raw(stmt(&self.schema, "INSERT INTO mst2_chunk_map_node(map_id,first_page,page_count,digest) SELECT $1,p.first_page,p.page_count,pg_catalog.decode(p.digest,'hex') FROM pg_catalog.jsonb_to_recordset($2::jsonb) AS p(first_page integer,page_count integer,digest text) ON CONFLICT(map_id,first_page,page_count) DO NOTHING", + [map.map_id().to_vec().into(), encoded.into()])).await.map_err(db_error)?; + } + } + compare_complete_map(&txn, &verified, &nodes, &self.schema).await?; + let fact = source.fact(); + txn.execute_raw(stmt(&self.schema, "INSERT INTO mst2_chunk_map_source(storage_domain,git_oid,object_kind,fact_id,source_id,source_bytes,primary_scope,map_id,receipt_digest) VALUES($1,$2,$3,$4,$5,$6,$7,$8,$9) ON CONFLICT(storage_domain,git_oid,object_kind) DO NOTHING", + [fact.storage_domain.clone().into(), fact.git_oid.clone().into(), fact.object_kind.clone().into(), fact.id.into(), + source_id.to_vec().into(), source_bytes.into(), self.primary_scope.clone().into(), map.map_id().to_vec().into(), Sha256::digest(&bytes).to_vec().into()])).await.map_err(db_error)?; + let row = source_row(&txn, source, &self.schema).await?.ok_or_else(|| integrity("installed source receipt is missing"))?; + let (stored, _) = self.validate_source_row(source, &row)?; + if stored != *map { return Err(integrity("concurrent source installation conflicts")); } + Ok(()) + }.await; + finish(txn, result).await + } + + pub(crate) async fn selected_page( + &self, + map: &PersistedChunkMap, + page_index: u64, + ) -> Result, SnapshotError> { + let intervals = proof_intervals(map.map.page_count, page_index)?; + let memory = self.budget.reserve(PAGE_CREDIT)?; + let txn = self.transaction().await?; + let result = async { + self.require_scope(&txn).await?; + require_current_source(&txn, &map.source, &self.schema).await?; + let row = txn.query_one_raw(stmt(&self.schema, "SELECT pg_catalog.octet_length(payload) AS size,CASE WHEN pg_catalog.octet_length(payload) BETWEEN 48 AND 8208 THEN payload ELSE NULL END AS payload FROM mst2_chunk_map_leaf WHERE map_id=$1 AND page_index=$2", + [map.map_id.to_vec().into(), (page_index as i32).into()])).await.map_err(db_error)? + .ok_or_else(|| integrity("admitted chunk map selected leaf is missing"))?; + let bytes = row.try_get::>>("", "payload").map_err(db_error)? + .ok_or_else(|| integrity("admitted chunk map selected leaf exceeds its byte profile"))?; + let leaf = ChunkLeaf::decode(&bytes).map_err(|_| integrity("admitted chunk map selected leaf is not canonical MCL2"))?; + if leaf.page_index != page_index || leaf.chunk_sha256.len() as u64 != ChunkLeaf::expected_count(map.map.chunk_count, page_index) + || leaf.encode().map_err(|_| integrity("invalid selected leaf"))? != bytes { + return Err(integrity("selected chunk map leaf identity or chunk count disagrees")); + } + let requested: Vec<_> = intervals.iter().map(|&(_, start, pages)| json!({"first_page":start,"page_count":pages})).collect(); + let encoded = serde_json::to_string(&requested).map_err(db_error)?; + let rows = txn.query_all_raw(stmt(&self.schema, "SELECT n.first_page,n.page_count,CASE WHEN pg_catalog.octet_length(n.digest)=32 THEN n.digest ELSE NULL END AS digest FROM pg_catalog.jsonb_to_recordset($2::jsonb) AS p(first_page integer,page_count integer) JOIN mst2_chunk_map_node n ON n.map_id=$1 AND n.first_page=p.first_page AND n.page_count=p.page_count", + [map.map_id.to_vec().into(), encoded.into()])).await.map_err(db_error)?; + let mut nodes = Vec::with_capacity(rows.len()); + for row in rows { + let digest: Option> = row.try_get("", "digest").map_err(db_error)?; + nodes.push(ChunkMapNode { + start: row.try_get::("", "first_page").map_err(db_error)? as u64, + pages: row.try_get::("", "page_count").map_err(db_error)? as u64, + digest: digest.ok_or_else(|| integrity("admitted chunk map sibling has invalid digest size"))?.as_slice().try_into().map_err(|_| integrity("invalid persisted sibling digest"))?, + }); + } + let proof = selected_proof(&intervals, &nodes)?; + verify_leaf(map.map.page_count, page_index, leaf.leaf_hash().map_err(|_| integrity("invalid leaf digest"))?, &proof, map.map.pages_root) + .map_err(|_| integrity("selected chunk map leaf or proof authentication failed"))?; + tracing::debug!(target:"mst2::chunk_map", selected_leaf_bytes = bytes.len(), selected_sibling_rows = nodes.len(), page_count = map.map.page_count, "authenticated persisted chunk map selected page"); + Ok(Arc::new(AuthenticatedChunkPage { leaf, proof, _memory: memory })) + }.await; + finish(txn, result).await + } + + async fn transaction(&self) -> Result { + let txn = self + .connection + .begin_with_config(Some(IsolationLevel::ReadCommitted), None) + .await + .map_err(db_error)?; + let search_path = format!("{},pg_catalog,pg_temp", quoted(&self.schema)); + txn.query_one_raw(Statement::from_sql_and_values( + DbBackend::Postgres, + "SELECT pg_catalog.set_config('search_path',$1,true)", + [search_path.into()], + )) + .await + .map_err(db_error)?; + txn.execute_unprepared("SET LOCAL statement_timeout='5s'; SET LOCAL lock_timeout='5s'") + .await + .map_err(db_error)?; + Ok(txn) + } + + async fn require_scope(&self, connection: &C) -> Result<(), SnapshotError> { + if read_scope(connection, &self.schema).await? != self.primary_scope { + return Err(integrity( + "chunk map repository no longer targets its captured primary scope", + )); + } + Ok(()) + } + + pub(crate) fn memory_budget(&self) -> &Arc { + &self.budget + } + + pub(crate) fn source_identity( + &self, + source: &ChunkMapSource, + ) -> Result<[u8; 32], SnapshotError> { + Ok(source_id(&self.primary_scope, &source.canonical_bytes()?)) + } + + #[cfg(test)] + pub(crate) fn with_test_budget(mut self, budget: Arc) -> Self { + self.budget = budget; + self + } + + #[cfg(test)] + pub(crate) fn test_primary_scope(&self) -> &[u8] { + &self.primary_scope + } + + fn validate_source_row( + &self, + source: &ChunkMapSource, + row: &QueryResult, + ) -> Result<(ChunkMap, [u8; 32]), SnapshotError> { + let source_bytes = source.canonical_bytes()?; + let source_id = source_id(&self.primary_scope, &source_bytes); + let source_bytes_row = bounded_bytes(row, "source_bytes")?; + let primary_scope_row = bounded_bytes(row, "primary_scope")?; + if row.try_get::("", "fact_id").map_err(db_error)? != source.fact().id + || source_bytes_row != source_bytes + || primary_scope_row != self.primary_scope + || bounded_bytes(row, "source_id")? != source_id + { + return Err(integrity( + "persisted map admission disagrees with the exact current source fact or primary", + )); + } + let bytes = row + .try_get::>>("", "descriptor") + .map_err(db_error)? + .ok_or_else(|| { + integrity("admitted chunk map descriptor is missing or outside its byte profile") + })?; + let map = ChunkMap::decode(&bytes) + .map_err(|_| integrity("admitted chunk map descriptor is invalid MCM2"))?; + let receipt_digest: [u8; 32] = + Sha256::digest(receipt(&self.primary_scope, source, &map)?).into(); + if map.encode() != bytes + || map.map_id().as_slice() != bounded_bytes(row, "map_id")?.as_slice() + || map.file_content_id.as_slice() != source.fact().raw_sha256.as_slice() + || map.file_size != source.fact().size as u64 + || map.page_count != row.try_get::("", "page_count").map_err(db_error)? as u64 + || map.pages_root.as_slice() != bounded_bytes(row, "pages_root")?.as_slice() + || bounded_bytes(row, "receipt_digest")? != receipt_digest + { + return Err(integrity( + "admitted chunk map descriptor or receipt identity disagrees", + )); + } + Ok((map, source_id)) + } +} + +async fn capture_schema(connection: &C) -> Result { + let row = connection.query_one_raw(Statement::from_string(DbBackend::Postgres, + "SELECT n.nspname AS schema FROM pg_catalog.pg_namespace n JOIN pg_catalog.pg_class c ON c.relnamespace=n.oid AND c.relname='mst2_metadata_storage_scope' AND c.relkind='r' WHERE n.nspname=pg_catalog.current_schema() AND n.nspname NOT LIKE 'pg_temp_%'")) + .await.map_err(db_error)?.ok_or_else(|| integrity("chunk map actual primary storage relation is missing"))?; + row.try_get("", "schema").map_err(db_error) +} + +async fn read_scope( + connection: &C, + schema: &str, +) -> Result, SnapshotError> { + if connection.get_database_backend() != DbBackend::Postgres { + return Err(integrity("chunk map repository requires PostgreSQL")); + } + let row = connection.query_one_raw(stmt(schema, "SELECT pg_catalog.pg_is_in_recovery() AS replica,CASE WHEN pg_catalog.octet_length(s.storage_uuid)<=255 THEN s.storage_uuid ELSE NULL END AS storage_uuid,pg_catalog.current_database() AS database,d.oid::bigint AS database_oid,n.nspname AS schema,n.oid::bigint AS schema_oid,pg_catalog.inet_server_addr()::text AS address,pg_catalog.inet_server_port() AS port FROM mst2_metadata_storage_scope s JOIN pg_catalog.pg_database d ON d.datname=pg_catalog.current_database() JOIN pg_catalog.pg_namespace n ON n.nspname=$1 JOIN pg_catalog.pg_class c ON c.relnamespace=n.oid AND c.relname='mst2_metadata_storage_scope' AND c.relkind='r' WHERE s.singleton=1", [schema.into()])) + .await.map_err(db_error)?.ok_or_else(|| integrity("chunk map primary scope is missing"))?; + if row.try_get::("", "replica").map_err(db_error)? { + return Err(integrity("chunk map repository is on a replica")); + } + let scope = serde_json::to_vec(&( + row.try_get::>("", "storage_uuid") + .map_err(db_error)? + .ok_or_else(|| integrity("chunk map primary scope UUID exceeds its bounded profile"))?, + row.try_get::("", "database").map_err(db_error)?, + row.try_get::("", "database_oid").map_err(db_error)?, + row.try_get::("", "schema").map_err(db_error)?, + row.try_get::("", "schema_oid").map_err(db_error)?, + row.try_get::>("", "address") + .map_err(db_error)?, + row.try_get::>("", "port").map_err(db_error)?, + )) + .map_err(db_error)?; + if scope.len() > 1024 { + return Err(integrity( + "chunk map primary scope exceeds its admitted profile", + )); + } + Ok(scope) +} + +async fn require_current_source( + connection: &C, + source: &ChunkMapSource, + schema: &str, +) -> Result<(), SnapshotError> { + let row = connection + .query_one_raw(stmt(schema, + "SELECT id,CASE WHEN storage_domain='git' THEN storage_domain ELSE NULL END AS storage_domain,CASE WHEN git_oid=$2 AND pg_catalog.octet_length(git_oid) IN (40,64) THEN git_oid ELSE NULL END AS git_oid,CASE WHEN object_kind='blob' THEN object_kind ELSE NULL END AS object_kind,CASE WHEN pg_catalog.octet_length(raw_sha256)=32 THEN raw_sha256 ELSE NULL END AS raw_sha256,CASE WHEN size BETWEEN 1 AND 8796093022208 THEN size ELSE NULL END AS size,CASE WHEN verification_version=2 THEN verification_version ELSE NULL END AS verification_version,CASE WHEN state='VERIFIED' THEN state ELSE NULL END AS state,created_at FROM mst2_verified_object WHERE id=$1 FOR SHARE", + [source.fact().id.into(), source.fact().git_oid.clone().into()], + )) + .await + .map_err(db_error)? + .ok_or_else(|| { + SnapshotError::new( + SnapshotErrorCode::MetadataNotReady, + "chunk map source has no current verified fact", + ) + })?; + let current = mst2_verified_object::Model::from_query_result(&row, "").map_err(|_| { + integrity("chunk map current source fact is outside its bounded verified profile") + })?; + if ¤t != source.fact() { + return Err(integrity( + "chunk map source fact changed during request or verification", + )); + } + Ok(()) +} + +async fn source_row( + connection: &C, + source: &ChunkMapSource, + schema: &str, +) -> Result, SnapshotError> { + connection.query_one_raw(stmt(schema, "SELECT s.fact_id,CASE WHEN pg_catalog.octet_length(s.source_id)=32 THEN s.source_id ELSE NULL END AS source_id,CASE WHEN pg_catalog.octet_length(s.source_bytes)<=2048 THEN s.source_bytes ELSE NULL END AS source_bytes,CASE WHEN pg_catalog.octet_length(s.primary_scope)<=1024 THEN s.primary_scope ELSE NULL END AS primary_scope,CASE WHEN pg_catalog.octet_length(s.map_id)=32 THEN s.map_id ELSE NULL END AS map_id,CASE WHEN pg_catalog.octet_length(s.receipt_digest)=32 THEN s.receipt_digest ELSE NULL END AS receipt_digest,CASE WHEN pg_catalog.octet_length(m.descriptor)=100 THEN m.descriptor ELSE NULL END AS descriptor,m.page_count,CASE WHEN pg_catalog.octet_length(m.pages_root)=32 THEN m.pages_root ELSE NULL END AS pages_root FROM mst2_chunk_map_source s LEFT JOIN mst2_chunk_map m ON m.map_id=s.map_id WHERE s.storage_domain=$1 AND s.git_oid=$2 AND s.object_kind=$3", + [source.fact().storage_domain.clone().into(), source.fact().git_oid.clone().into(), source.fact().object_kind.clone().into()])).await.map_err(db_error) +} + +fn source_id(scope: &[u8], source: &[u8]) -> [u8; 32] { + let mut hash = Sha256::new(); + hash.update(RECEIPT_DOMAIN); + hash.update((scope.len() as u32).to_be_bytes()); + hash.update(scope); + hash.update((source.len() as u32).to_be_bytes()); + hash.update(source); + hash.finalize().into() +} + +fn receipt( + scope: &[u8], + source: &ChunkMapSource, + map: &ChunkMap, +) -> Result, SnapshotError> { + let source = source.canonical_bytes()?; + if scope.len() > 1024 || source.len() > 2048 { + return Err(integrity( + "chunk map source receipt exceeds its fixed profile", + )); + } + let mut bytes = Vec::with_capacity(RECEIPT_DOMAIN.len() + 8 + scope.len() + source.len() + 100); + bytes.extend_from_slice(RECEIPT_DOMAIN); + bytes.extend_from_slice(&(scope.len() as u32).to_be_bytes()); + bytes.extend_from_slice(scope); + bytes.extend_from_slice(&(source.len() as u32).to_be_bytes()); + bytes.extend_from_slice(&source); + bytes.extend_from_slice(&map.encode()); + Ok(bytes) +} + +fn receipt_key(source_id: [u8; 32]) -> ObjectKey { + ObjectKey { + namespace: ObjectNamespace::ChunkMapReceipt, + key: hex::encode(source_id), + } +} + +async fn read_receipt( + objects: &MegaObjectStorageWrapper, + key: &ObjectKey, + expected: &[u8], +) -> Result<(), SnapshotError> { + tokio::time::timeout(std::time::Duration::from_secs(5), async { + let (mut stream, meta) = objects + .inner + .get_stream(key) + .await + .map_err(|_| integrity("admitted chunk map trusted receipt is unavailable"))?; + if meta.size != expected.len() as i64 { + return Err(integrity( + "chunk map trusted receipt has invalid total size", + )); + } + let mut offset = 0; + while let Some(part) = stream.next().await { + let bytes = part.map_err(|_| integrity("chunk map trusted receipt stream failed"))?; + if bytes.len() > expected.len() - offset + || &expected[offset..offset + bytes.len()] != bytes.as_ref() + { + return Err(integrity( + "chunk map trusted receipt bytes disagree with source and descriptor", + )); + } + offset += bytes.len(); + } + if offset != expected.len() { + return Err(integrity("chunk map trusted receipt is truncated")); + } + Ok(()) + }) + .await + .map_err(|_| { + SnapshotError::new( + SnapshotErrorCode::TemporaryUnavailable, + "chunk map receipt read timed out", + ) + })? +} + +fn encode_leaves(leaves: &[ChunkLeaf]) -> Result { + let values: Result, SnapshotError> = leaves.iter().map(|l| Ok(json!({"page_index":l.page_index,"payload":hex::encode(l.encode().map_err(|_| integrity("invalid verified leaf"))?)}))).collect(); + serde_json::to_string(&values?).map_err(db_error) +} + +fn encode_nodes(nodes: &[ChunkMapNode]) -> Result { + serde_json::to_string(&nodes.iter().map(|n| json!({"first_page":n.start,"page_count":n.pages,"digest":hex::encode(n.digest)})).collect::>()).map_err(db_error) +} + +async fn compare_complete_map( + txn: &DatabaseTransaction, + verified: &VerifiedSourceChunkMap, + nodes: &[ChunkMapNode], + schema: &str, +) -> Result<(), SnapshotError> { + let map = verified.map(); + let row = txn.query_one_raw(stmt(schema, "SELECT CASE WHEN pg_catalog.octet_length(descriptor)=100 THEN descriptor ELSE NULL END AS descriptor,page_count,CASE WHEN pg_catalog.octet_length(pages_root)=32 THEN pages_root ELSE NULL END AS pages_root,(SELECT pg_catalog.count(*) FROM mst2_chunk_map_leaf WHERE map_id=$1) AS leaves,(SELECT pg_catalog.count(*) FROM mst2_chunk_map_node WHERE map_id=$1) AS nodes FROM mst2_chunk_map WHERE map_id=$1", [map.map_id().to_vec().into()])) + .await.map_err(db_error)?.ok_or_else(|| integrity("chunk map installation descriptor is missing"))?; + if bounded_bytes(&row, "descriptor")? != map.encode() + || row.try_get::("", "page_count").map_err(db_error)? as u64 != map.page_count + || bounded_bytes(&row, "pages_root")? != map.pages_root + || row.try_get::("", "leaves").map_err(db_error)? as u64 != map.page_count + || row.try_get::("", "nodes").map_err(db_error)? as usize != nodes.len() + { + return Err(integrity( + "chunk map installation has conflicting descriptor or incomplete coverage", + )); + } + for batch in verified.leaves().chunks(64) { + let bad = txn.query_one_raw(stmt(schema, "SELECT p.page_index FROM pg_catalog.jsonb_to_recordset($2::jsonb) AS p(page_index integer,payload text) LEFT JOIN mst2_chunk_map_leaf l ON l.map_id=$1 AND l.page_index=p.page_index WHERE l.payload IS DISTINCT FROM pg_catalog.decode(p.payload,'hex') LIMIT 1", + [map.map_id().to_vec().into(), encode_leaves(batch)?.into()])).await.map_err(db_error)?; + if bad.is_some() { + return Err(integrity( + "immutable stored chunk map leaf conflicts with verified source", + )); + } + } + for batch in nodes.chunks(1024) { + let bad = txn.query_one_raw(stmt(schema, "SELECT p.first_page FROM pg_catalog.jsonb_to_recordset($2::jsonb) AS p(first_page integer,page_count integer,digest text) LEFT JOIN mst2_chunk_map_node n ON n.map_id=$1 AND n.first_page=p.first_page AND n.page_count=p.page_count WHERE n.digest IS DISTINCT FROM pg_catalog.decode(p.digest,'hex') LIMIT 1", + [map.map_id().to_vec().into(), encode_nodes(batch)?.into()])).await.map_err(db_error)?; + if bad.is_some() { + return Err(integrity( + "immutable stored chunk map node conflicts with verified source", + )); + } + } + Ok(()) +} + +fn quoted(schema: &str) -> String { + format!("\"{}\"", schema.replace('"', "\"\"")) +} + +/// The static statements use these exact relation tokens. Catalog functions +/// are explicitly qualified; relations never resolve through pg_temp. +fn stmt(schema: &str, sql: &str, values: [sea_orm::Value; N]) -> Statement { + Statement::from_sql_and_values(DbBackend::Postgres, qualify_relations(schema, sql), values) +} + +fn qualify_relations(schema: &str, sql: &str) -> String { + let mut qualified = String::with_capacity(sql.len() + 256); + let prefix = quoted(schema); + let bytes = sql.as_bytes(); + let mut cursor = 0; + while cursor < bytes.len() { + let start = cursor; + let byte = bytes[cursor]; + if byte == b'\'' || byte == b'"' { + cursor += 1; + while cursor < bytes.len() { + if bytes[cursor] == byte { + cursor += 1; + if cursor < bytes.len() && bytes[cursor] == byte { + cursor += 1; + } else { + break; + } + } else { + cursor += 1; + } + } + } else if byte.is_ascii_alphanumeric() || byte == b'_' { + cursor += 1; + while cursor < bytes.len() + && (bytes[cursor].is_ascii_alphanumeric() || bytes[cursor] == b'_') + { + cursor += 1; + } + if [ + "mst2_metadata_storage_scope", + "mst2_verified_object", + "mst2_chunk_map", + "mst2_chunk_map_source", + "mst2_chunk_map_leaf", + "mst2_chunk_map_node", + ] + .contains(&&sql[start..cursor]) + { + qualified.push_str(&prefix); + qualified.push('.'); + } + } else { + cursor += sql[cursor..].chars().next().map_or(1, char::len_utf8); + } + qualified.push_str(&sql[start..cursor]); + } + qualified +} + +#[cfg(test)] +mod sql_tests { + use super::*; + + #[test] + fn authority_relations_are_qualified_without_rewriting_catalog_literals() { + let sql = "SELECT 'mst2_metadata_storage_scope','quoted ''mst2_chunk_map''',\"mst2_chunk_map\" FROM mst2_metadata_storage_scope s JOIN mst2_verified_object f ON s.singleton=1 JOIN mst2_chunk_map m ON true JOIN mst2_chunk_map_leaf l ON true JOIN mst2_chunk_map_node n ON true JOIN mst2_chunk_map_source r ON true WHERE c.relname='mst2_metadata_storage_scope'"; + let qualified = qualify_relations("schema\"name", sql); + assert!(qualified.contains(&format!( + "FROM {}.mst2_metadata_storage_scope", + quoted("schema\"name") + ))); + for name in [ + "mst2_verified_object", + "mst2_chunk_map", + "mst2_chunk_map_leaf", + "mst2_chunk_map_node", + "mst2_chunk_map_source", + ] { + assert!(qualified.contains(&format!("JOIN {}.{name}", quoted("schema\"name")))); + } + assert!(qualified.starts_with( + "SELECT 'mst2_metadata_storage_scope','quoted ''mst2_chunk_map''',\"mst2_chunk_map\"" + )); + assert!(qualified.ends_with("WHERE c.relname='mst2_metadata_storage_scope'")); + } +} +fn bounded_bytes(row: &QueryResult, name: &str) -> Result, SnapshotError> { + row.try_get::>>("", name) + .map_err(db_error)? + .ok_or_else(|| { + integrity("persisted chunk map field is missing or outside its bounded profile") + }) +} +fn integrity(message: &str) -> SnapshotError { + SnapshotError::new(SnapshotErrorCode::IntegrityError, message) +} +fn db_error(error: impl std::fmt::Display) -> SnapshotError { + tracing::warn!(%error, "persisted chunk map storage operation failed"); + SnapshotError::new( + SnapshotErrorCode::TemporaryUnavailable, + "persisted chunk map storage operation failed", + ) +} +fn storage_error(error: impl std::fmt::Display) -> SnapshotError { + tracing::warn!(%error, "trusted chunk map receipt publication failed"); + SnapshotError::new( + SnapshotErrorCode::TemporaryUnavailable, + "trusted chunk map receipt publication failed", + ) +} +async fn finish( + txn: DatabaseTransaction, + result: Result, +) -> Result { + match result { + Ok(value) => { + txn.commit().await.map_err(db_error)?; + Ok(value) + } + Err(error) => { + let _ = txn.rollback().await; + Err(error) + } + } +} + +impl super::Storage { + pub(crate) async fn chunk_maps(&self) -> Result<&PostgresChunkMapRepository, SnapshotError> { + use super::base_storage::StorageConnector; + self.native_chunk_maps + .get_or_try_init(|| async { + PostgresChunkMapRepository::new(self.mono_storage().get_connection().clone()).await + }) + .await + } +} diff --git a/src/jupiter/storage/object_storage.rs b/src/jupiter/storage/object_storage.rs index 506f2974..f65a33ab 100644 --- a/src/jupiter/storage/object_storage.rs +++ b/src/jupiter/storage/object_storage.rs @@ -78,6 +78,110 @@ mod exact_range_tests { ); } } + + #[tokio::test] + async fn memory_chunk_receipts_reject_all_mutation_routes() { + let storage = mock_object_storage(); + super::assert_immutable_chunk_receipt_contract(storage.inner.as_ref()).await; + } +} + +#[cfg(test)] +pub(crate) async fn assert_immutable_chunk_receipt_contract( + storage: &dyn crate::orbit_api::factory::MegaObjectStorageWithLog, +) { + let key = ObjectKey { + namespace: crate::orbit_api::object_storage::ObjectNamespace::ChunkMapReceipt, + key: "a".repeat(64), + }; + let first = Bytes::from_static(b"trusted immutable receipt"); + storage + .put_metadata_atomic_create(&key, first.clone(), ObjectMeta::default()) + .await + .unwrap(); + storage + .put_metadata_atomic_create( + &key, + Bytes::from_static(b"conflicting replay"), + ObjectMeta::default(), + ) + .await + .unwrap(); + let input = || { + Box::pin(futures::stream::iter([Ok(Bytes::from_static( + b"overwrite", + ))])) as ObjectByteStream + }; + assert!( + storage + .put_stream(&key, input(), ObjectMeta::default()) + .await + .is_err() + ); + assert!( + storage + .put_stream_bounded(&key, input(), ObjectMeta::default()) + .await + .is_err() + ); + assert!( + storage + .put_metadata_atomic( + &key, + Bytes::from_static(b"overwrite"), + ObjectMeta::default() + ) + .await + .is_err() + ); + assert!( + storage + .put_metadata_atomic_create( + &key, + Bytes::from(vec![ + 0; + crate::orbit_api::object_storage::MAX_METADATA_ATOMIC_BYTES + + 1 + ]), + ObjectMeta::default() + ) + .await + .is_err() + ); + assert!( + storage + .append(&key, input(), ObjectMeta::default()) + .await + .is_err() + ); + assert!( + storage + .append_concurrently(&key, input(), ObjectMeta::default()) + .await + .is_err() + ); + assert!(storage.delete(&key).await.is_err()); + for method in [Method::PUT, Method::POST, Method::DELETE, Method::PATCH] { + assert!( + storage + .signed_url(&key, method, std::time::Duration::from_secs(60)) + .await + .is_err() + ); + } + assert!( + storage + .signed_url(&key, Method::GET, std::time::Duration::from_secs(60)) + .await + .is_ok() + ); + let (mut stream, meta) = storage.get_stream(&key).await.unwrap(); + assert_eq!(meta.size, first.len() as i64); + let mut observed = Vec::new(); + while let Some(part) = stream.next().await { + observed.extend_from_slice(&part.unwrap()); + } + assert_eq!(observed, first.as_ref()); } #[derive(Default)] @@ -111,6 +215,7 @@ impl MegaObjectStorage for InMemoryObjectStorage { data: ObjectByteStream, mut meta: ObjectMeta, ) -> OrbitResult<()> { + reject_receipt_mutation(key)?; let bytes = Self::read_stream(data).await?; meta.size = bytes.len() as i64; self.objects @@ -126,6 +231,7 @@ impl MegaObjectStorage for InMemoryObjectStorage { bytes: Bytes, mut meta: ObjectMeta, ) -> OrbitResult<()> { + reject_receipt_mutation(key)?; use crate::orbit_api::object_storage::MAX_METADATA_ATOMIC_BYTES; if bytes.len() > MAX_METADATA_ATOMIC_BYTES { return Err(IoOrbitError::Other(format!( @@ -140,6 +246,27 @@ impl MegaObjectStorage for InMemoryObjectStorage { Ok(()) } + async fn put_metadata_atomic_create( + &self, + key: &ObjectKey, + bytes: Bytes, + mut meta: ObjectMeta, + ) -> OrbitResult<()> { + if bytes.len() > crate::orbit_api::object_storage::MAX_METADATA_ATOMIC_BYTES { + return Err(IoOrbitError::Other( + "immutable atomic metadata exceeds its byte limit".into(), + )); + } + key.validate()?; + meta.size = bytes.len() as i64; + self.objects + .lock() + .map_err(|_| IoOrbitError::Other("object storage lock poisoned".into()))? + .entry(key.clone()) + .or_insert((bytes, meta)); + Ok(()) + } + async fn get_stream(&self, key: &ObjectKey) -> OrbitResult<(ObjectByteStream, ObjectMeta)> { let (bytes, meta) = self .objects @@ -208,14 +335,18 @@ impl MegaObjectStorage for InMemoryObjectStorage { async fn signed_url( &self, - _key: &ObjectKey, - _method: Method, + key: &ObjectKey, + method: Method, _expires_in: std::time::Duration, ) -> OrbitResult> { + if method != Method::GET { + reject_receipt_mutation(key)?; + } Ok(None) } async fn delete(&self, key: &ObjectKey) -> OrbitResult<()> { + reject_receipt_mutation(key)?; let deleted = self .objects .lock() @@ -230,6 +361,16 @@ impl MegaObjectStorage for InMemoryObjectStorage { } } +fn reject_receipt_mutation(key: &ObjectKey) -> OrbitResult<()> { + if key.namespace == crate::orbit_api::object_storage::ObjectNamespace::ChunkMapReceipt { + Err(IoOrbitError::Other( + "chunk map receipts require immutable atomic creation".into(), + )) + } else { + Ok(()) + } +} + #[async_trait::async_trait] impl LogStorage for InMemoryObjectStorage { async fn append( @@ -238,6 +379,7 @@ impl LogStorage for InMemoryObjectStorage { data: ObjectByteStream, mut meta: ObjectMeta, ) -> OrbitResult<()> { + reject_receipt_mutation(key)?; let bytes = Self::read_stream(data).await?; let mut objects = self .objects diff --git a/src/orbit/adapter/log.rs b/src/orbit/adapter/log.rs index 4d116744..6dd1d78f 100644 --- a/src/orbit/adapter/log.rs +++ b/src/orbit/adapter/log.rs @@ -8,6 +8,7 @@ impl LogStorage for ObjectStoreAdapter { data: ObjectByteStream, _meta: ObjectMeta, ) -> OrbitResult<()> { + super::object::reject_receipt_mutation(key)?; key.validate()?; // Fast path for single writer/single thread (optimized here): // - No conditional writes, no retries, no cleanup (no concurrent write conflicts) @@ -187,6 +188,7 @@ impl LogStorage for ObjectStoreAdapter { data: ObjectByteStream, _meta: ObjectMeta, ) -> OrbitResult<()> { + super::object::reject_receipt_mutation(key)?; key.validate()?; // The local backend has no conditional-write (CAS) support, so this // method cannot be made safe under contention there (it would fall back diff --git a/src/orbit/adapter/object.rs b/src/orbit/adapter/object.rs index 58ccacc0..87327a39 100644 --- a/src/orbit/adapter/object.rs +++ b/src/orbit/adapter/object.rs @@ -12,6 +12,7 @@ impl MegaObjectStorage for ObjectStoreAdapter { data: ObjectByteStream, _meta: ObjectMeta, ) -> OrbitResult<()> { + reject_receipt_mutation(key)?; let path = Self::checked_path(key)?; // Artifacts are keyed by UUID and must not depend on LFS/Git upload_strategy // (e.g. S3 often uses `Multipart` for LFS while we still need create-if-absent semantics). @@ -35,6 +36,7 @@ impl MegaObjectStorage for ObjectStoreAdapter { data: ObjectByteStream, _meta: ObjectMeta, ) -> OrbitResult<()> { + reject_receipt_mutation(key)?; let path = Self::checked_path(key)?; self.put_multipart(&path, data).await } @@ -45,6 +47,7 @@ impl MegaObjectStorage for ObjectStoreAdapter { bytes: Bytes, _meta: ObjectMeta, ) -> OrbitResult<()> { + reject_receipt_mutation(key)?; use crate::orbit_api::object_storage::MAX_METADATA_ATOMIC_BYTES; if bytes.len() > MAX_METADATA_ATOMIC_BYTES { return Err(IoOrbitError::Other(format!( @@ -61,6 +64,32 @@ impl MegaObjectStorage for ObjectStoreAdapter { Ok(()) } + async fn put_metadata_atomic_create( + &self, + key: &ObjectKey, + bytes: Bytes, + _meta: ObjectMeta, + ) -> OrbitResult<()> { + if bytes.len() > crate::orbit_api::object_storage::MAX_METADATA_ATOMIC_BYTES { + return Err(IoOrbitError::Other( + "immutable atomic metadata exceeds its byte limit".into(), + )); + } + let path = Self::checked_path(key)?; + match self + .to_store() + .put_opts( + &path, + PutPayload::from_bytes(bytes), + PutOptions::from(PutMode::Create), + ) + .await + { + Ok(_) | Err(object_store::Error::AlreadyExists { .. }) => Ok(()), + Err(error) => Err(IoOrbitError::from(error)), + } + } + async fn get_stream(&self, key: &ObjectKey) -> OrbitResult<(ObjectByteStream, ObjectMeta)> { let path = Self::checked_path(key)?; @@ -149,6 +178,11 @@ impl MegaObjectStorage for ObjectStoreAdapter { method: Method, expires_in: Duration, ) -> OrbitResult> { + if key.namespace == ObjectNamespace::ChunkMapReceipt && method != Method::GET { + return Err(IoOrbitError::Other( + "immutable chunk map receipts do not permit presigned mutation".into(), + )); + } let path = Self::checked_path(key)?; let url = match &self.store { @@ -178,6 +212,7 @@ impl MegaObjectStorage for ObjectStoreAdapter { } async fn delete(&self, key: &ObjectKey) -> OrbitResult<()> { + reject_receipt_mutation(key)?; let path = Self::checked_path(key)?; self.to_store() .delete(&path) @@ -187,10 +222,34 @@ impl MegaObjectStorage for ObjectStoreAdapter { } } +pub(super) fn reject_receipt_mutation(key: &ObjectKey) -> OrbitResult<()> { + if key.namespace == ObjectNamespace::ChunkMapReceipt { + Err(IoOrbitError::Other( + "chunk map receipts require immutable atomic creation".into(), + )) + } else { + Ok(()) + } +} + #[cfg(test)] mod exact_range_tests { use super::*; + #[tokio::test] + async fn local_chunk_receipts_reject_all_mutation_routes() { + let directory = tempfile::tempdir().unwrap(); + let adapter = ObjectStoreAdapter { + store: BackendStore::Local(Arc::new( + LocalFileSystem::new_with_prefix(directory.path()).unwrap(), + )), + upload_strategy: UploadStrategy::SinglePut, + presign_store: None, + }; + crate::jupiter::storage::object_storage::assert_immutable_chunk_receipt_contract(&adapter) + .await; + } + #[tokio::test] async fn local_exact_range_returns_selected_bytes_and_full_object_size_without_clipping() { let directory = tempfile::tempdir().unwrap(); diff --git a/src/orbit_api/object_storage.rs b/src/orbit_api/object_storage.rs index 7fe6f47a..90539315 100644 --- a/src/orbit_api/object_storage.rs +++ b/src/orbit_api/object_storage.rs @@ -122,6 +122,8 @@ pub enum ObjectNamespace { Media, /// Agent Capture objects (`docs/refactoring/agent-capture.md`). Agent, + /// Immutable receipts minted by the full-stream chunk-map verifier. + ChunkMapReceipt, } impl ObjectNamespace { @@ -135,6 +137,7 @@ impl ObjectNamespace { ObjectNamespace::Oci => "oci", ObjectNamespace::Media => "media", ObjectNamespace::Agent => "agent", + ObjectNamespace::ChunkMapReceipt => "chunk-map-receipt", } } } @@ -260,6 +263,21 @@ pub trait MegaObjectStorage: Send + Sync { )) } + /// Create a complete <=1 MiB immutable metadata object. An existing key + /// must retain its original bytes. The caller must read and compare the + /// existing object before treating an idempotent replay as success. + async fn put_metadata_atomic_create( + &self, + _key: &ObjectKey, + _bytes: Bytes, + _meta: ObjectMeta, + ) -> OrbitResult<()> { + Err(IoOrbitError::Other( + "immutable atomic metadata creation is not supported by this storage backend" + .to_string(), + )) + } + /// Retrieve a single object from the storage backend. /// /// # Returns @@ -527,6 +545,7 @@ mod tests { ObjectNamespace::Oci, ObjectNamespace::Media, ObjectNamespace::Agent, + ObjectNamespace::ChunkMapReceipt, ] { let key = ObjectKey { namespace: ns, @@ -550,6 +569,10 @@ mod tests { assert_eq!(ObjectNamespace::Oci.to_string(), "oci"); assert_eq!(ObjectNamespace::Media.to_string(), "media"); assert_eq!(ObjectNamespace::Agent.to_string(), "agent"); + assert_eq!( + ObjectNamespace::ChunkMapReceipt.to_string(), + "chunk-map-receipt" + ); } #[test]