From 1c0eb297d066dafff976e38c4ec9a6b5b523c918 Mon Sep 17 00:00:00 2001 From: Xiaoyang Han Date: Thu, 8 Oct 2026 04:47:47 +0800 Subject: [PATCH] perf(mst2): persist source-bound chunk maps and authenticate selected pages Actual chunk-map, page and CHUNK routes replace the digest-only process cache with immutable source-bound PostgreSQL indexes and independent object-store receipts minted only after one complete current-OID size/SHA256/chunk/EOF pass. Exact-source cold callers share an installation gate, then each request reads its own backend receipt. Reconstructed warm readers consume no full source body and fetch only selected MCL2 pages and exact Merkle sibling intervals; CHUNK still reads and authenticates the current request OID range. Captured physical primary scope, bounded qualified SQL, full fact tuple, opaque verifier publication, deferred complete atomic installation, mutation-resistant receipts, owned descriptor/page/JSON credits, cancellation and per-transport lease checks retain fail-closed behavior. Old cache/runtime and page alias are removed; MCM2/MCL2 schema_version=2 remains the formal wire codec within v3. PostgreSQL and actual HTTP regression fixtures cover forged ordinary-DML indexes, receipt size/bytes/EOF, corrupted selected proofs, source and primary changes, temp shadows, SQL rollback, concurrency, cancellation and owned-reader release. Source formatting/diff/offline metadata pass. Native compile, Clippy, PG/HTTP tests and measured latency/RSS/Git comparison remain pending. --- .../router/snapshot_chunks_bounded_tests.rs | 29 + src/api/router/snapshot_content.rs | 292 +++- src/api/router/snapshot_content_tests.rs | 161 +- .../router/snapshot_objects_bounded_tests.rs | 27 +- .../snapshot_persisted_chunk_map_tests.rs | 1303 +++++++++++++++++ src/ceres/snapshot/chunk_map_gate.rs | 185 +++ src/ceres/snapshot/chunk_map_index.rs | 171 +++ .../snapshot/chunk_singleflight_tests.rs | 487 ------ src/ceres/snapshot/chunks.rs | 357 ++--- src/ceres/snapshot/chunks_stream.rs | 116 +- src/ceres/snapshot/chunks_stream_tests.rs | 269 ++-- src/ceres/snapshot/mod.rs | 2 + .../m20261008_000100_add_mst2_chunk_maps.rs | 28 + .../migration/m20261008_000100_chunk_maps.sql | 90 ++ src/jupiter/migration/mod.rs | 2 + src/jupiter/storage/mod.rs | 5 + src/jupiter/storage/mono_storage.rs | 43 +- src/jupiter/storage/native_chunk_map.rs | 684 +++++++++ src/jupiter/storage/object_storage.rs | 146 +- src/orbit/adapter/log.rs | 2 + src/orbit/adapter/object.rs | 59 + src/orbit_api/object_storage.rs | 23 + 22 files changed, 3433 insertions(+), 1048 deletions(-) create mode 100644 src/api/router/snapshot_persisted_chunk_map_tests.rs create mode 100644 src/ceres/snapshot/chunk_map_gate.rs create mode 100644 src/ceres/snapshot/chunk_map_index.rs delete mode 100644 src/ceres/snapshot/chunk_singleflight_tests.rs create mode 100644 src/jupiter/migration/m20261008_000100_add_mst2_chunk_maps.rs create mode 100644 src/jupiter/migration/m20261008_000100_chunk_maps.sql create mode 100644 src/jupiter/storage/native_chunk_map.rs diff --git a/src/api/router/snapshot_chunks_bounded_tests.rs b/src/api/router/snapshot_chunks_bounded_tests.rs index 8d22a68d..243d326a 100644 --- a/src/api/router/snapshot_chunks_bounded_tests.rs +++ b/src/api/router/snapshot_chunks_bounded_tests.rs @@ -214,6 +214,14 @@ async fn mst2_large_chunk_uses_current_oid_strict_range_faults_cancel_retry_and_ fixture.counts.assert(1, size as usize); let map_id = map["map"]["map_id"].as_str().unwrap(); fixture.counts.reset(); + super::persisted_chunk_maps::assert_three_page_proofs_and_selected_sibling_faults( + &fixture, map_id, digest, &pattern, + ) + .await; + // Each distinct OID earns its own full-stream receipt before map reuse. + assert_eq!(fixture.map("/other").await["map"], map["map"]); + fixture.counts.assert(1, size as usize); + fixture.counts.reset(); // Equal content map sharing never carries the first source's OID. let body = request("/other", digest, map_id, 512); assert_chunk( @@ -387,6 +395,26 @@ async fn mst2_chunk_batch_live_budget_and_invalid_later_path_reject_before_body_ ], ) .await; + let budget = crate::ceres::snapshot::content_budget::MemoryBudget::new(1024 * 1024); + let repository = crate::jupiter::storage::native_chunk_map::PostgresChunkMapRepository::new( + fixture + .state + .storage + .mono_storage() + .get_connection() + .clone(), + ) + .await + .unwrap() + .with_test_budget(budget.clone()); + assert!( + fixture + .state + .storage + .native_chunk_maps + .set(repository) + .is_ok() + ); let mut items = Vec::new(); for (index, path) in ["/file", "/one", "/two"].iter().enumerate() { let oid = oid_for(&fixture, path).await; @@ -417,6 +445,7 @@ async fn mst2_chunk_batch_live_budget_and_invalid_later_path_reject_before_body_ ) .await; fixture.counts.assert(0, 0); + assert_eq!(budget.used(), 0); let fixture = Fixture::new().await; let mut body = fixture.chunk_body("/file", &format!("sha256:{}", hex_of(&[1; 32])), "0"); let mut invalid = body["items"][0].clone(); diff --git a/src/api/router/snapshot_content.rs b/src/api/router/snapshot_content.rs index ea7d8938..d642434f 100644 --- a/src/api/router/snapshot_content.rs +++ b/src/api/router/snapshot_content.rs @@ -4,10 +4,9 @@ //! WP and `frame_encodings` advertises identity alone. use axum::{ - Json, extract::{Path as AxumPath, Query, State}, http::HeaderMap, - response::{IntoResponse, Response}, + response::Response, }; use futures::stream::StreamExt; use serde::Deserialize; @@ -19,8 +18,8 @@ use super::{ mst2_error_response, request::Mst2Bytes, }; use crate::ceres::snapshot::{ - chunks::{ChunkProjection, get_or_project_stream, projection_reservation_bytes}, - content_budget::{PROJECTION_LIVE_BYTES, reserve_range_work, reserve_response}, + chunks::{ChunkMapSource, VerifiedSourceChunkMap, map_build_reservation_bytes}, + content_budget::{BudgetedFrame, MemoryLease, reserve_range_work, reserve_response}, error::{SnapshotError, SnapshotErrorCode}, pages::{MetadataWalkOutcome, base64_of, hex_of, resolve_abs_metadata}, resolver::FsKind, @@ -32,6 +31,7 @@ pub(super) struct ResolvedFileMetadata { oid: String, pub(super) digest: [u8; 32], pub(super) size: u64, + fact: crate::callisto::mst2_verified_object::Model, } #[allow(clippy::result_large_err)] @@ -154,6 +154,7 @@ pub(super) async fn verified_file_metadata, #[serde(default)] - pub(super) page: Option, - /// Canonical v3 page requests bind the page to the already verified map - /// instead of repeating the file digest. - #[serde(default)] pub(super) map_id: Option, #[serde(default)] pub(super) page_index: Option, @@ -475,7 +472,8 @@ async fn project_for( scope: &str, path: &str, expected_digest: Option<&str>, -) -> Result, Response> { +) -> Result, Response> +{ let f = resolve_file_metadata(handler, root_tree, scope, path, expected_digest).await?; project_resolved(handler, &f).await } @@ -484,36 +482,61 @@ async fn project_for( async fn project_resolved( handler: &T, f: &ResolvedFileMetadata, -) -> Result, Response> { - // The first request for a digest builds the projection from the fixed - // Git object; later requests slice the cached representation. A miss - // rebuilds, never errors with "missing chunk". - let digest = f.digest; - let size = f.size; - let projection = get_or_project_stream(digest, size, || async move { - handler - .get_raw_blob_stream_by_hash(&f.oid) - .await - .map_err(content_read_error) - }) - .await - .map_err(|error| { - mst2_error_response(if error.code == SnapshotErrorCode::DigestMismatch { - SnapshotError::new( - SnapshotErrorCode::IntegrityError, - "fixed blob digest disagrees with its verified fact", - ) - } else { - error - }) - })?; - if projection.map.file_size != size || projection.map.file_content_id != digest { +) -> Result, Response> +{ + if f.size == 0 { return Err(mst2_error_response(SnapshotError::new( - SnapshotErrorCode::IntegrityError, - "cached projection disagrees with the fixed verified fact", + SnapshotErrorCode::ScopeInvalid, + "empty files have no chunk map", ))); } - Ok(projection) + let source = ChunkMapSource::from_fact(f.fact.clone(), &f.oid).map_err(mst2_error_response)?; + let storage = handler.get_context(); + let repository = storage.chunk_maps().await.map_err(mst2_error_response)?; + let objects = &storage.git_service.obj_storage; + if let Some(map) = repository + .read(&source, objects) + .await + .map_err(mst2_error_response)? + { + return Ok(map); + } + let flight = crate::ceres::snapshot::chunk_map_gate::InstallFlight::acquire( + repository + .source_identity(&source) + .map_err(mst2_error_response)?, + ) + .map_err(mst2_error_response)?; + let _gate = flight.lock().await.map_err(mst2_error_response)?; + if let Some(map) = repository + .read(&source, objects) + .await + .map_err(mst2_error_response)? + { + return Ok(map); + } + let verified = + VerifiedSourceChunkMap::verify(handler, source.clone(), repository.memory_budget()) + .await + .map_err(|error| { + mst2_error_response(if error.code == SnapshotErrorCode::DigestMismatch { + SnapshotError::new( + SnapshotErrorCode::IntegrityError, + "fixed blob digest disagrees with its verified fact", + ) + } else { + error + }) + })?; + repository + .install(verified, objects) + .await + .map_err(mst2_error_response)?; + repository + .read(&source, objects) + .await + .map_err(mst2_error_response)? + .ok_or_else(|| mst2_error_response(internal("installed chunk map source is missing"))) } #[allow(clippy::result_large_err)] @@ -539,10 +562,18 @@ pub(super) async fn chunk_map( q.expected_digest.as_deref(), ) .await?; - // Canonical v3 uses a closed top-level envelope and a nested map - // descriptor. Keeping the map under `map` is part of profile selection; - // the client rejects the legacy flat shape once canonical capabilities - // have been advertised. + let wire_bound = q + .path + .len() + .checked_mul(6) + .and_then(|n| n.checked_add(4096)) + .ok_or_else(|| mst2_error_response(internal("chunk map response bound overflow")))?; + let memory = reserve_response( + wire_bound + .checked_mul(2) + .ok_or_else(|| mst2_error_response(internal("chunk map response credit overflow")))?, + ) + .map_err(mst2_error_response)?; let body = json!({ "snapshot_id": snapshot_id, "path": q.path, @@ -557,7 +588,9 @@ pub(super) async fn chunk_map( "map_id": format!("sha256:{}", hex_of(&proj.map_id)), }, }); - Ok(Json(body).into_response()) + guarded_map_json_response(&state, &ctx, &body, memory) + .await + .map_err(mst2_error_response) } #[allow(clippy::result_large_err)] @@ -569,19 +602,13 @@ pub(super) async fn chunk_map_pages( ensure(&state)?; let ctx = super::request_context(&state, &snapshot_id).map_err(mst2_error_response)?; validate_scope_relative_path(&q.path).map_err(mst2_error_response)?; - // Canonical v3 uses `map_id` + `page_index`; retain parsing of the old - // names only while the legacy client is still present in this checkout. - let page_index: u64 = match (q.page_index.as_deref(), q.page.as_deref()) { - (Some(_), Some(_)) => { - return Err(mst2_error_response(SnapshotError::new( - SnapshotErrorCode::ScopeInvalid, - "page_index and page are mutually exclusive", - ))); - } - (Some(s), None) => parse_decimal_count(s, "page_index").map_err(mst2_error_response)?, - (None, Some(s)) => parse_decimal_count(s, "page").map_err(mst2_error_response)?, - (None, None) => 0, - }; + let page_index = q + .page_index + .as_deref() + .map(|s| parse_decimal_count(s, "page_index")) + .transpose() + .map_err(mst2_error_response)? + .unwrap_or(0); if q.page_index.is_some() != q.map_id.is_some() { return Err(mst2_error_response(SnapshotError::new( SnapshotErrorCode::ScopeInvalid, @@ -611,9 +638,19 @@ pub(super) async fn chunk_map_pages( ))); } } - let (leaf, proof) = proj - .leaf_and_proof(page_index) + let storage = handler.get_context(); + let page = storage + .chunk_maps() + .await + .map_err(mst2_error_response)? + .selected_page(&proj, page_index) + .await .map_err(mst2_error_response)?; + // Reserve both JSON values and the encoded wire buffer before either + // allocation. The authenticated leaf/proof retain their own lease. + let memory = reserve_response(64 * 1024).map_err(mst2_error_response)?; + let leaf = &page.leaf; + let proof = &page.proof; let leaf_bytes = leaf .encode() .map_err(|e| mst2_error_response(internal(format!("chunk leaf encode failed: {e}"))))?; @@ -639,7 +676,84 @@ pub(super) async fn chunk_map_pages( "leaf_base64": base64_of(&leaf_bytes), "proof": proof_json, }); - Ok(Json(body).into_response()) + guarded_map_json_response(&state, &ctx, &body, memory) + .await + .map_err(mst2_error_response) +} + +fn map_json_bytes( + value: &serde_json::Value, + memory: MemoryLease, +) -> Result { + let mut bytes = Vec::new(); + bytes + .try_reserve_exact(memory.bytes / 2) + .map_err(|_| internal("chunk map JSON allocation failed"))?; + let limit = memory.bytes / 2; + let mut writer = BoundedMapJsonWriter { + bytes: &mut bytes, + limit, + }; + serde_json::to_writer(&mut writer, value) + .map_err(|_| internal("chunk map JSON encoding exceeds its owned credit"))?; + if bytes.capacity() > memory.bytes { + return Err(internal( + "chunk map JSON allocation exceeds its owned credit", + )); + } + Ok(bytes::Bytes::from_owner(BudgetedFrame { + bytes, + lease: std::sync::Arc::new(memory), + })) +} + +struct BoundedMapJsonWriter<'a> { + bytes: &'a mut Vec, + limit: usize, +} + +impl std::io::Write for BoundedMapJsonWriter<'_> { + fn write(&mut self, input: &[u8]) -> std::io::Result { + if input.len() > self.limit - self.bytes.len() { + return Err(std::io::Error::other( + "chunk map JSON exceeds its wire bound", + )); + } + self.bytes.extend_from_slice(input); + Ok(input.len()) + } + + fn flush(&mut self) -> std::io::Result<()> { + Ok(()) + } +} + +async fn guarded_map_json_response( + state: &crate::api::MonoApiServiceState, + context: &crate::ceres::snapshot::runtime::SnapshotContext, + value: &serde_json::Value, + memory: MemoryLease, +) -> Result { + let bytes = map_json_bytes(value, memory)?; + let headers = super::REQUEST_HEADERS.try_with(Clone::clone).map_err(|_| { + SnapshotError::new( + SnapshotErrorCode::Unauthenticated, + "request authentication context missing", + ) + })?; + super::revalidate_access(state, context, &headers).await?; + let state = state.clone(); + let context = context.clone(); + let stream = futures::stream::once(async move { + super::revalidate_access(&state, &context, &headers).await?; + Ok::<_, SnapshotError>(bytes) + }); + Response::builder() + .header("content-type", "application/json") + .header("cache-control", "private, no-cache, no-transform") + .header("vary", "Authorization, Accept") + .body(axum::body::Body::from_stream(stream)) + .map_err(|_| internal("chunk map JSON response build failed")) } #[derive(Deserialize, Debug)] @@ -663,7 +777,8 @@ const CHUNKS_MAX_ITEMS: usize = 128; const CHUNKS_TOTAL_MAX: u64 = 128 * 1024 * 1024; struct Planned { - projection: std::sync::Arc, + projection: std::sync::Arc, + page: std::sync::Arc, oid: String, index: u64, } @@ -743,8 +858,9 @@ async fn read_chunk_range( } raw.extend_from_slice(&bytes); } - projection - .verify_chunk(planned.index, &raw) + planned + .page + .verify_chunk(&projection.map, planned.index, &raw) .map_err(mst2_error_response)?; Ok(raw) } @@ -784,7 +900,6 @@ pub(super) async fn chunks( let mut resolved: Vec = Vec::new(); let mut units: Vec<(String, u64)> = Vec::new(); let mut distinct: std::collections::HashMap<[u8; 32], u64> = std::collections::HashMap::new(); - let mut projection_bytes = 0usize; let mut logical_bytes = 0u64; for item in &req.items { validate_scope_relative_path(&item.path).map_err(mst2_error_response)?; @@ -836,16 +951,10 @@ pub(super) async fn chunks( } } std::collections::hash_map::Entry::Vacant(entry) => { - let bytes = projection_reservation_bytes(file.size).map_err(mst2_error_response)?; - projection_bytes = projection_bytes.checked_add(bytes).ok_or_else(|| { - mst2_error_response(internal("projection batch memory overflow")) - })?; - if projection_bytes > PROJECTION_LIVE_BYTES { - return Err(mst2_error_response(SnapshotError::new( - SnapshotErrorCode::LimitExceeded, - "chunk batch exceeds its live projection memory budget", - ))); - } + // Check the profile before any body I/O. Cold construction + // owns its credits and is dropped after durable installation; + // warm requests retain only descriptors and selected pages. + map_build_reservation_bytes(file.size).map_err(mst2_error_response)?; entry.insert(file.size); } } @@ -860,8 +969,19 @@ pub(super) async fn chunks( .and_then(|bytes| bytes.checked_add(req.items.len() * 1024 + 1024)) .ok_or_else(|| mst2_error_response(internal("chunk response memory overflow")))?; let response_memory = reserve_response(response_bytes).map_err(mst2_error_response)?; + let mut maps = std::collections::HashMap::new(); + let mut pages = std::collections::HashMap::new(); for item in resolved { - let proj = project_resolved(handler.as_ref(), &item.file).await?; + let source = ChunkMapSource::from_fact(item.file.fact.clone(), &item.file.oid) + .map_err(mst2_error_response)?; + let source_key = source.canonical_bytes().map_err(mst2_error_response)?; + let proj = if let Some(map) = maps.get(&source_key) { + std::sync::Arc::clone(map) + } else { + let map = project_resolved(handler.as_ref(), &item.file).await?; + maps.insert(source_key, std::sync::Arc::clone(&map)); + map + }; let want_map = format!("sha256:{}", hex_of(&proj.map_id)); if want_map != item.map_id { return Err(mst2_error_response(SnapshotError::new( @@ -869,8 +989,27 @@ pub(super) async fn chunks( "map_id does not bind to the fixed file", ))); } + let page_key = ( + proj.source_id(), + item.index / mst2_codec::chunkmap::CHUNKS_PER_PAGE as u64, + ); + let page = if let Some(page) = pages.get(&page_key) { + std::sync::Arc::clone(page) + } else { + let storage = handler.get_context(); + let page = storage + .chunk_maps() + .await + .map_err(mst2_error_response)? + .selected_page(&proj, page_key.1) + .await + .map_err(mst2_error_response)?; + pages.insert(page_key, std::sync::Arc::clone(&page)); + page + }; planned.push(Planned { projection: proj, + page, oid: item.file.oid, index: item.index, }); @@ -883,14 +1022,7 @@ pub(super) async fn chunks( let _work_memory = reserve_range_work().map_err(mst2_error_response)?; // The source is the current request's fixed OID, never a cached // handler/backend/credential from a different scope. - let bytes = if p.projection.has_inline_bytes() { - p.projection - .chunk_bytes(p.index) - .map_err(mst2_error_response)? - .to_vec() - } else { - read_chunk_range(handler.as_ref(), p).await? - }; + let bytes = read_chunk_range(handler.as_ref(), p).await?; let frame = stream .chunk( p.projection.map_id, diff --git a/src/api/router/snapshot_content_tests.rs b/src/api/router/snapshot_content_tests.rs index bb58d788..4399e692 100644 --- a/src/api/router/snapshot_content_tests.rs +++ b/src/api/router/snapshot_content_tests.rs @@ -1,7 +1,7 @@ use std::{ sync::{ Arc, - atomic::{AtomicUsize, Ordering}, + atomic::{AtomicBool, AtomicUsize, Ordering}, }, time::Duration, }; @@ -85,6 +85,9 @@ mod bounded_objects; #[path = "snapshot_chunks_bounded_tests.rs"] mod bounded_chunks; +#[path = "snapshot_persisted_chunk_map_tests.rs"] +mod persisted_chunk_maps; + #[path = "snapshot_session_tests.rs"] mod durable_sessions; @@ -100,6 +103,8 @@ mod generation_qualified_fixture; #[path = "snapshot_install_capability_fixture.rs"] mod install_capability_fixture; +type ReceiptWriteHold = (Arc, Arc); + #[derive(Default)] struct ReadCounts { whole: AtomicUsize, @@ -107,6 +112,14 @@ struct ReadCounts { bytes: AtomicUsize, object_fault: std::sync::Mutex>, chunk_faults: std::sync::Mutex>, + receipt_reads: AtomicUsize, + receipt_writes: AtomicUsize, + receipt_write_fail_after_create: AtomicBool, + receipt_read_failure: AtomicBool, + receipt_read_corruption: std::sync::Mutex>, + receipt_read_meta_size: std::sync::atomic::AtomicI64, + receipt_read_late_error: AtomicBool, + receipt_write_holds: std::sync::Mutex>, } impl ReadCounts { @@ -114,6 +127,8 @@ impl ReadCounts { self.whole.store(0, Ordering::SeqCst); self.range.store(0, Ordering::SeqCst); self.bytes.store(0, Ordering::SeqCst); + self.receipt_reads.store(0, Ordering::SeqCst); + self.receipt_writes.store(0, Ordering::SeqCst); } fn assert(&self, whole: usize, bytes: usize) { @@ -130,6 +145,36 @@ struct CountingStorage { #[async_trait::async_trait] impl MegaObjectStorage for CountingStorage { + async fn put_metadata_atomic_create( + &self, + key: &ObjectKey, + bytes: Bytes, + meta: ObjectMeta, + ) -> OrbitResult<()> { + self.inner + .inner + .put_metadata_atomic_create(key, bytes, meta) + .await?; + if key.namespace == ObjectNamespace::ChunkMapReceipt { + self.counts.receipt_writes.fetch_add(1, Ordering::SeqCst); + let holds = self.counts.receipt_write_holds.lock().unwrap().clone(); + if let Some((entered, release)) = holds { + entered.notify_one(); + release.notified().await; + } + if self + .counts + .receipt_write_fail_after_create + .swap(false, Ordering::SeqCst) + { + return Err(crate::orbit_api::error::IoOrbitError::Other( + "injected post-create receipt failure".into(), + )); + } + } + Ok(()) + } + async fn put_stream( &self, key: &ObjectKey, @@ -140,6 +185,39 @@ impl MegaObjectStorage for CountingStorage { } async fn get_stream(&self, key: &ObjectKey) -> OrbitResult<(ObjectByteStream, ObjectMeta)> { + if key.namespace == ObjectNamespace::ChunkMapReceipt { + self.counts.receipt_reads.fetch_add(1, Ordering::SeqCst); + if self.counts.receipt_read_failure.load(Ordering::SeqCst) { + return Err( + crate::orbit_api::error::IoOrbitError::object_store_not_found( + key.default_sharding(), + ), + ); + } + let bad = self.counts.receipt_read_corruption.lock().unwrap().clone(); + if let Some(bytes) = bad { + let declared = self.counts.receipt_read_meta_size.load(Ordering::SeqCst); + let size = if declared > 0 { + declared + } else { + bytes.len() as i64 + }; + let mut parts = vec![Ok(bytes)]; + if self.counts.receipt_read_late_error.load(Ordering::SeqCst) { + parts.push(Err(std::io::Error::other( + "injected late receipt read error", + ))); + } + return Ok(( + Box::pin(futures::stream::iter(parts)), + ObjectMeta { + size, + ..Default::default() + }, + )); + } + return self.inner.inner.get_stream(key).await; + } self.counts.whole.fetch_add(1, Ordering::SeqCst); let chunk_fault = self .counts @@ -220,10 +298,21 @@ impl MegaObjectStorage for CountingStorage { (Box::pin(stream) as ObjectByteStream, meta) })); } - self.inner + let result = self + .inner .inner .get_range_stream_exact(key, start, end) - .await + .await?; + Ok(result.map(|(stream, meta)| { + let counts = self.counts.clone(); + let stream = stream.map(move |part| { + if let Ok(bytes) = &part { + counts.bytes.fetch_add(bytes.len(), Ordering::SeqCst); + } + part + }); + (Box::pin(stream) as ObjectByteStream, meta) + })) } async fn exists(&self, key: &ObjectKey) -> OrbitResult { @@ -500,11 +589,16 @@ impl Fixture { .save_object_from_raw(Bytes::copy_from_slice(raw)) .await .unwrap(); - project_items.push(item( - TreeItemMode::Blob, - ObjectHash::from_hex_for_kind(HashKind::Sha1, &oid).unwrap(), - name, - )); + let mut oid = ObjectHash::from_hex_for_kind(HashKind::Sha1, &oid).unwrap(); + let mut mode = TreeItemMode::Blob; + let components: Vec<_> = name.split('/').collect(); + for index in (1..components.len()).rev() { + let child = tree(vec![item(mode, oid, components[index])]); + oid = child.id; + mode = TreeItemMode::Tree; + extra_trees.push(child); + } + project_items.push(item(mode, oid, components[0])); } let project = tree(project_items); let old_tip = Commit::from_tree_id_with_kind( @@ -814,7 +908,7 @@ async fn mst2_fixed_head_uses_verified_facts_without_body_reads_and_preserves_ra } #[tokio::test] -async fn mst2_fixed_warm_map_leaf_and_chunk_aliases_skip_body_reads() { +async fn mst2_fixed_warm_map_and_leaf_aliases_skip_body_reads_and_chunks_use_current_ranges() { let fixture = Fixture::new().await; let initial = fixture.map("/file").await; fixture.counts.assert(1, fixture.raw.len()); @@ -823,18 +917,6 @@ async fn mst2_fixed_warm_map_leaf_and_chunk_aliases_skip_body_reads() { assert_eq!(initial["map"]["chunk_count"], "2"); let map_id = initial["map"]["map_id"].as_str().unwrap(); fixture.counts.reset(); - fixture - .state - .storage - .git_service - .obj_storage - .inner - .delete(&ObjectKey { - namespace: ObjectNamespace::Git, - key: fixture.oid.clone(), - }) - .await - .unwrap(); for path in ["/file", "/alias", "/executable", "/nested/file"] { let cached = fixture.map(path).await; assert_eq!(cached["path"], path); @@ -902,8 +984,43 @@ async fn mst2_fixed_warm_map_leaf_and_chunk_aliases_skip_body_reads() { <[u8; 32]>::from(Sha256::digest(body.as_bytes())) ); } - fixture.counts.assert(0, 0); + assert_eq!(fixture.counts.whole.load(Ordering::SeqCst), 0); + assert_eq!(fixture.counts.range.load(Ordering::SeqCst), 2); + assert_eq!( + fixture.counts.bytes.load(Ordering::SeqCst), + fixture.raw.len() + ); + fixture.counts.reset(); } + fixture + .state + .storage + .git_service + .obj_storage + .inner + .delete(&ObjectKey { + namespace: ObjectNamespace::Git, + key: fixture.oid.clone(), + }) + .await + .unwrap(); + assert_eq!(fixture.map("/alias").await["map"], initial["map"]); + fixture.counts.assert(0, 0); + error( + fixture + .send( + "POST", + "chunks", + Body::from(fixture.chunk_body("/alias", map_id, "0").to_string()), + ) + .await, + 503, + "OBJECT_UNAVAILABLE", + false, + ) + .await; + assert_eq!(fixture.counts.whole.load(Ordering::SeqCst), 0); + assert_eq!(fixture.counts.range.load(Ordering::SeqCst), 1); } #[tokio::test] diff --git a/src/api/router/snapshot_objects_bounded_tests.rs b/src/api/router/snapshot_objects_bounded_tests.rs index d001de12..b5e0d303 100644 --- a/src/api/router/snapshot_objects_bounded_tests.rs +++ b/src/api/router/snapshot_objects_bounded_tests.rs @@ -7,11 +7,11 @@ use super::*; #[derive(Clone)] pub(super) struct StreamFault { pub(super) oid: String, - kind: FaultKind, + pub(super) kind: FaultKind, } #[derive(Clone)] -enum FaultKind { +pub(super) enum FaultKind { Parts(Vec), LateError(Bytes), Oversized(Bytes, Arc), @@ -21,6 +21,11 @@ enum FaultKind { release: Arc, drops: Arc, }, + HeldThenError { + entered: Arc, + release: Arc, + drops: Arc, + }, } struct DropCount(Arc); @@ -34,6 +39,24 @@ impl Drop for DropCount { impl StreamFault { pub(super) fn stream(self) -> ObjectByteStream { match self.kind { + FaultKind::HeldThenError { + entered, + release, + drops, + } => Box::pin(futures::stream::unfold( + (true, entered, release, DropCount(drops)), + |(first, entered, release, owner)| async move { + if !first { + return None; + } + entered.notify_one(); + release.notified().await; + Some(( + Err(io::Error::other("held source read failed")), + (false, entered, release, owner), + )) + }, + )), FaultKind::Parts(parts) => Box::pin(futures::stream::iter(parts.into_iter().map(Ok))), FaultKind::LateError(raw) => Box::pin(futures::stream::iter([ Ok(raw), diff --git a/src/api/router/snapshot_persisted_chunk_map_tests.rs b/src/api/router/snapshot_persisted_chunk_map_tests.rs new file mode 100644 index 00000000..6b261115 --- /dev/null +++ b/src/api/router/snapshot_persisted_chunk_map_tests.rs @@ -0,0 +1,1303 @@ +use sea_orm::{DatabaseConnection, DbBackend, IsolationLevel, Statement, TransactionTrait}; +use tokio::{sync::Notify, time::timeout}; + +use super::*; +use crate::{ + ceres::snapshot::{ + chunks::{ChunkMapSource, ChunkProjection}, + content_budget::MemoryBudget, + }, + jupiter::storage::native_chunk_map::PostgresChunkMapRepository, +}; + +fn statement(sql: &str, values: [sea_orm::Value; N]) -> Statement { + Statement::from_sql_and_values(DbBackend::Postgres, sql, values) +} + +async fn count(db: &DatabaseConnection, table: &str) -> i64 { + db.query_one_raw(statement( + &format!("SELECT count(*) AS count FROM {table}"), + [], + )) + .await + .unwrap() + .unwrap() + .try_get("", "count") + .unwrap() +} + +async fn reconstructed(fixture: &Fixture) -> MonoApiServiceState { + let mut config = (*fixture.state.storage.config()).clone(); + config.database.max_connection = 1; + config.database.min_connection = 1; + let config = Arc::new(config); + let connection = crate::jupiter::storage::init::postgres_connection(&config.database) + .await + .unwrap(); + let storage = crate::jupiter::storage::Storage::new_with_connection( + config, + Arc::new(connection), + fixture.state.storage.git_service.obj_storage.clone(), + ) + .await + .unwrap(); + MonoApiServiceState { + storage, + git_object_cache: Arc::new(GitObjectCache { + connection: fixture.state.git_object_cache.connection.clone(), + prefix: uuid::Uuid::new_v4().to_string(), + }), + ..fixture.state.clone() + } +} + +fn router(state: &MonoApiServiceState) -> Router { + Router::new().nest("/api/v2", routers(state.clone()).with_state(state.clone())) +} + +#[tokio::test] +async fn persisted_map_rebuilt_actual_http_uses_canonical_pages_and_only_requested_raw_range() { + let fixture = Fixture::new_with_pg_config(true).await; + let original = fixture.map("/file").await; + fixture.counts.assert(1, fixture.raw.len()); + assert_eq!(fixture.counts.receipt_writes.load(Ordering::SeqCst), 1); + let oracle = ChunkProjection::build(fixture.digest, fixture.raw.clone()).unwrap(); + assert_eq!( + original["map"]["map_id"], + format!("sha256:{}", hex_of(&oracle.map_id)) + ); + let state = reconstructed(&fixture).await; + assert!(state.storage.native_chunk_maps.get().is_none()); + let app = router(&state); + fixture.counts.reset(); + let map = success_json( + app.clone() + .oneshot(fixture.request("GET", "chunk-map?path=/alias", Body::empty())) + .await + .unwrap(), + ) + .await; + assert_eq!(map["map"], original["map"]); + let map_id = map["map"]["map_id"].as_str().unwrap(); + let page = success_json( + app.clone() + .oneshot(fixture.request( + "GET", + &format!("chunk-map/pages?path=/alias&map_id={map_id}&page_index=0"), + Body::empty(), + )) + .await + .unwrap(), + ) + .await; + let bytes = STANDARD + .decode(page["leaf_base64"].as_str().unwrap()) + .unwrap(); + let (leaf, proof) = oracle.leaf_and_proof(0).unwrap(); + assert_eq!(bytes, leaf.encode().unwrap()); + assert!(proof.is_empty()); + assert_eq!(page["proof"], json!([])); + fixture.counts.assert(0, 0); + assert!(fixture.counts.receipt_reads.load(Ordering::SeqCst) >= 2); + assert_eq!(fixture.counts.receipt_writes.load(Ordering::SeqCst), 0); + let body = fixture.chunk_body("/alias", map_id, "1").to_string(); + let response = app + .oneshot(fixture.request("POST", "chunks", Body::from(body.clone()))) + .await + .unwrap(); + assert_eq!(response.status(), 200); + let wire = to_bytes(response.into_body(), 2 * 1024 * 1024) + .await + .unwrap(); + let frames = parse_stream(&wire).unwrap(); + let [Frame::Chunk(chunk), Frame::End(end)] = frames.as_slice() else { + panic!("expected exact CHUNK and END"); + }; + assert_eq!(chunk.chunk_bytes, fixture.raw[CHUNK_SIZE as usize..]); + assert_eq!(chunk.chunk_index, 1); + assert_eq!(chunk.file_content_id, fixture.digest); + assert_eq!(chunk.map_id, oracle.map_id); + assert_eq!(end.request_item_count, 1); + assert_eq!(end.unique_unit_count, 1); + assert_eq!(end.logical_bytes, 113); + assert_eq!( + end.request_body_sha256, + <[u8; 32]>::from(Sha256::digest(body.as_bytes())) + ); + assert_eq!(fixture.counts.whole.load(Ordering::SeqCst), 0); + assert_eq!(fixture.counts.range.load(Ordering::SeqCst), 1); + assert_eq!(fixture.counts.bytes.load(Ordering::SeqCst), 113); +} + +#[tokio::test] +async fn persisted_map_current_fact_tuple_receipt_and_leaf_corruption_fail_closed_without_fallback() +{ + let fixture = Fixture::new().await; + let map = fixture.map("/file").await; + let original = fixture.fact().await; + let map_id = map["map"]["map_id"].as_str().unwrap(); + fixture.counts.reset(); + for case in 0..3 { + let mut fact = original.clone(); + match case { + 0 => fact.id += 1_000_000, + 1 => fact.created_at += chrono::Duration::seconds(1), + _ => fact.raw_sha256[0] ^= 1, + } + fixture.replace_fact(fact).await; + error( + fixture + .send("GET", "chunk-map?path=/file", Body::empty()) + .await, + 502, + "INTEGRITY_ERROR", + false, + ) + .await; + fixture.counts.assert(0, 0); + } + fixture.replace_fact(original).await; + fixture + .counts + .receipt_read_failure + .store(true, Ordering::SeqCst); + error( + fixture + .send("GET", "chunk-map?path=/file", Body::empty()) + .await, + 502, + "INTEGRITY_ERROR", + false, + ) + .await; + fixture + .counts + .receipt_read_failure + .store(false, Ordering::SeqCst); + *fixture.counts.receipt_read_corruption.lock().unwrap() = + Some(Bytes::from_static(b"forged DB proof")); + error( + fixture + .send("GET", "chunk-map?path=/file", Body::empty()) + .await, + 502, + "INTEGRITY_ERROR", + false, + ) + .await; + *fixture.counts.receipt_read_corruption.lock().unwrap() = None; + let mono = fixture.state.storage.mono_storage(); + let db = mono.get_connection(); + let leaf: Vec = db + .query_one_raw(statement( + "SELECT payload FROM mst2_chunk_map_leaf WHERE map_id=$1 AND page_index=0", + [hex::decode(map_id.trim_start_matches("sha256:")) + .unwrap() + .into()], + )) + .await + .unwrap() + .unwrap() + .try_get("", "payload") + .unwrap(); + db.execute_unprepared("ALTER TABLE mst2_chunk_map_leaf DISABLE TRIGGER USER") + .await + .unwrap(); + let mut bad = leaf.clone(); + bad[16] ^= 1; + db.execute_raw(statement( + "UPDATE mst2_chunk_map_leaf SET payload=$1", + [bad.into()], + )) + .await + .unwrap(); + db.execute_unprepared("ALTER TABLE mst2_chunk_map_leaf ENABLE TRIGGER USER") + .await + .unwrap(); + for (method, suffix, body) in [ + ( + "GET", + format!("chunk-map/pages?path=/file&map_id={map_id}&page_index=0"), + Body::empty(), + ), + ( + "POST", + "chunks".to_string(), + Body::from(fixture.chunk_body("/file", map_id, "0").to_string()), + ), + ] { + error( + fixture.send(method, &suffix, body).await, + 502, + "INTEGRITY_ERROR", + false, + ) + .await; + } + fixture.counts.assert(0, 0); + db.execute_unprepared("ALTER TABLE mst2_chunk_map_leaf DISABLE TRIGGER USER") + .await + .unwrap(); + db.execute_raw(statement( + "UPDATE mst2_chunk_map_leaf SET payload=$1", + [leaf.into()], + )) + .await + .unwrap(); + db.execute_unprepared("ALTER TABLE mst2_chunk_map_leaf ENABLE TRIGGER USER") + .await + .unwrap(); + assert_eq!(fixture.map("/file").await["map"], map["map"]); + fixture.counts.assert(0, 0); +} + +#[tokio::test] +async fn ordinary_db_dml_cannot_forge_full_body_admission_or_mutate_admitted_indexes() { + let fixture = Fixture::new().await; + let source = ChunkMapSource::from_fact(fixture.fact().await, &fixture.oid).unwrap(); + let mono = fixture.state.storage.mono_storage(); + let db = mono.get_connection(); + let repository = PostgresChunkMapRepository::new(db.clone()).await.unwrap(); + let scope = repository.test_primary_scope(); + let source_bytes = source.canonical_bytes().unwrap(); + let leaf = ChunkLeaf { + page_index: 0, + chunk_sha256: vec![[9; 32]; 2], + }; + let root = leaf.leaf_hash().unwrap(); + let map = mst2_codec::chunkmap::ChunkMap::new(fixture.digest, fixture.raw.len() as u64, root) + .unwrap(); + let mut receipt = b"MST2-CHUNK-MAP-RECEIPT\0".to_vec(); + receipt.extend_from_slice(&(scope.len() as u32).to_be_bytes()); + receipt.extend_from_slice(scope); + receipt.extend_from_slice(&(source_bytes.len() as u32).to_be_bytes()); + receipt.extend_from_slice(&source_bytes); + let source_id: [u8; 32] = Sha256::digest(&receipt).into(); + receipt.extend_from_slice(&map.encode()); + let txn = db + .begin_with_config(Some(IsolationLevel::ReadCommitted), None) + .await + .unwrap(); + txn.execute_raw(statement( + "INSERT INTO mst2_chunk_map(map_id,descriptor,page_count,pages_root) VALUES($1,$2,1,$3)", + [ + map.map_id().to_vec().into(), + map.encode().into(), + root.to_vec().into(), + ], + )) + .await + .unwrap(); + txn.execute_raw(statement( + "INSERT INTO mst2_chunk_map_leaf(map_id,page_index,payload) VALUES($1,0,$2)", + [map.map_id().to_vec().into(), leaf.encode().unwrap().into()], + )) + .await + .unwrap(); + txn.execute_raw(statement( + "INSERT INTO mst2_chunk_map_node(map_id,first_page,page_count,digest) VALUES($1,0,1,$2)", + [map.map_id().to_vec().into(), root.to_vec().into()], + )) + .await + .unwrap(); + txn.execute_raw(statement("INSERT INTO mst2_chunk_map_source(storage_domain,git_oid,object_kind,fact_id,source_id,source_bytes,primary_scope,map_id,receipt_digest) VALUES('git',$1,'blob',$2,$3,$4,$5,$6,$7)", [fixture.oid.clone().into(),source.fact().id.into(),source_id.to_vec().into(),source_bytes.into(),scope.to_vec().into(),map.map_id().to_vec().into(),Sha256::digest(&receipt).to_vec().into()])).await.unwrap(); + txn.commit().await.unwrap(); + // All row checks and even a self-computed receipt digest pass. They + // still cannot create the trusted writer's independent object receipt. + fixture.counts.reset(); + error( + fixture + .send("GET", "chunk-map?path=/file", Body::empty()) + .await, + 502, + "INTEGRITY_ERROR", + false, + ) + .await; + fixture.counts.assert(0, 0); + assert_eq!(fixture.counts.receipt_writes.load(Ordering::SeqCst), 0); + for table in [ + "mst2_chunk_map", + "mst2_chunk_map_leaf", + "mst2_chunk_map_node", + "mst2_chunk_map_source", + ] { + assert!( + db.execute_unprepared(&format!("DELETE FROM {table}")) + .await + .is_err() + ); + assert!( + db.execute_unprepared(&format!("TRUNCATE {table} CASCADE")) + .await + .is_err() + ); + } + assert!( + db.execute_unprepared("UPDATE mst2_chunk_map_source SET fact_id=fact_id") + .await + .is_err() + ); + assert!(db.execute_raw(statement("INSERT INTO mst2_chunk_map_node(map_id,first_page,page_count,digest) VALUES($1,1,1,$2)", [map.map_id().to_vec().into(),root.to_vec().into()])).await.is_err()); +} + +#[tokio::test] +async fn receipt_orphan_failure_and_cancelled_install_release_owned_credit_and_replay_atomically() { + for cancel in [false, true] { + let fixture = Fixture::new().await; + let mono = fixture.state.storage.mono_storage(); + let budget = MemoryBudget::new(8 * 1024 * 1024); + let repository = PostgresChunkMapRepository::new(mono.get_connection().clone()) + .await + .unwrap() + .with_test_budget(budget.clone()); + assert!( + fixture + .state + .storage + .native_chunk_maps + .set(repository) + .is_ok() + ); + if cancel { + let entered = Arc::new(Notify::new()); + let release = Arc::new(Notify::new()); + *fixture.counts.receipt_write_holds.lock().unwrap() = Some((entered.clone(), release)); + let app = fixture.app.clone(); + let request = fixture.request("GET", "chunk-map?path=/file", Body::empty()); + let task = tokio::spawn(async move { app.oneshot(request).await }); + timeout(Duration::from_secs(10), entered.notified()) + .await + .unwrap(); + assert!(budget.used() > 4 * 1024 * 1024); + assert_eq!( + count(mono.get_connection(), "mst2_chunk_map_source").await, + 0 + ); + task.abort(); + assert!(task.await.err().unwrap().is_cancelled()); + *fixture.counts.receipt_write_holds.lock().unwrap() = None; + } else { + fixture + .counts + .receipt_write_fail_after_create + .store(true, Ordering::SeqCst); + error( + fixture + .send("GET", "chunk-map?path=/file", Body::empty()) + .await, + 503, + "TEMPORARY_UNAVAILABLE", + true, + ) + .await; + fixture + .counts + .receipt_write_fail_after_create + .store(false, Ordering::SeqCst); + } + assert_eq!(budget.used(), 0); + fixture.counts.assert(1, fixture.raw.len()); + assert_eq!(fixture.counts.receipt_writes.load(Ordering::SeqCst), 1); + assert_eq!(count(mono.get_connection(), "mst2_chunk_map").await, 0); + fixture.counts.reset(); + fixture.map("/file").await; + fixture.counts.assert(1, fixture.raw.len()); + assert_eq!(fixture.counts.receipt_writes.load(Ordering::SeqCst), 1); + assert_eq!(count(mono.get_connection(), "mst2_chunk_map").await, 1); + assert_eq!( + count(mono.get_connection(), "mst2_chunk_map_source").await, + 1 + ); + assert_eq!(count(mono.get_connection(), "mst2_chunk_map_leaf").await, 1); + assert_eq!(count(mono.get_connection(), "mst2_chunk_map_node").await, 1); + fixture.counts.reset(); + fixture.map("/file").await; + fixture.counts.assert(0, 0); + assert_eq!(fixture.counts.receipt_writes.load(Ordering::SeqCst), 0); + assert_eq!(budget.used(), 0); + } +} + +#[tokio::test] +async fn map_json_transport_bytes_keep_owned_credit_until_the_last_clone_drops() { + let budget = MemoryBudget::new(4096); + let bytes = super::super::map_json_bytes( + &json!({"map_id":"sha256:owned"}), + budget.reserve(4096).unwrap(), + ) + .unwrap(); + let mut stream = Body::from(bytes).into_data_stream(); + let transport = stream.next().await.unwrap().unwrap(); + let clone = transport.clone(); + drop(transport); + drop(stream); + assert_eq!(budget.used(), 4096); + assert!(budget.reserve(1).is_err()); + drop(clone); + assert_eq!(budget.used(), 0); + assert!(budget.reserve(4096).is_ok()); +} + +#[test] +fn json_wire_limit_rejects_growth_and_refunds_credit() { + let budget = MemoryBudget::new(4096); + let value = json!({"path":"\u{1}".repeat(1000)}); + let error = super::super::map_json_bytes(&value, budget.reserve(4096).unwrap()) + .err() + .unwrap(); + assert_eq!(error.code, SnapshotErrorCode::Internal); + assert_eq!(budget.used(), 0); +} + +async fn observer(fixture: &Fixture) -> crate::ceres::snapshot::chunk_map_gate::InstallFlight { + let repository = fixture.state.storage.chunk_maps().await.unwrap(); + let source = ChunkMapSource::from_fact(fixture.fact().await, &fixture.oid).unwrap(); + crate::ceres::snapshot::chunk_map_gate::InstallFlight::acquire( + repository.source_identity(&source).unwrap(), + ) + .unwrap() +} + +async fn wait_owners( + flight: &crate::ceres::snapshot::chunk_map_gate::InstallFlight, + owners: usize, +) { + timeout(Duration::from_secs(10), async { + loop { + if flight.test_owner_count() == owners { + break; + } + tokio::task::yield_now().await; + } + }) + .await + .expect("actual HTTP callers did not reach the same-source install gate"); +} + +fn held_leader(fixture: &Fixture) -> (Arc, Arc, tokio::task::JoinHandle) { + let entered = Arc::new(Notify::new()); + let release = Arc::new(Notify::new()); + *fixture.counts.receipt_write_holds.lock().unwrap() = Some((entered.clone(), release.clone())); + let app = fixture.app.clone(); + let request = fixture.request("GET", "chunk-map?path=/file", Body::empty()); + let task = tokio::spawn(async move { app.oneshot(request).await.unwrap() }); + (entered, release, task) +} + +#[tokio::test] +async fn same_source_cold_actual_http_callers_share_one_full_pass_and_each_recheck_their_receipt() { + let fixture = Fixture::new().await; + let (entered, release, leader) = held_leader(&fixture); + timeout(Duration::from_secs(10), entered.notified()) + .await + .unwrap(); + let flight = observer(&fixture).await; + let mut joined = Vec::new(); + for _ in 0..6 { + let app = fixture.app.clone(); + let request = fixture.request("GET", "chunk-map?path=/alias", Body::empty()); + joined.push(tokio::spawn( + async move { app.oneshot(request).await.unwrap() }, + )); + } + let other_counts = Arc::new(ReadCounts::default()); + other_counts + .receipt_read_failure + .store(true, Ordering::SeqCst); + let mut other_state = fixture.state.clone(); + other_state.storage.git_service = GitService { + obj_storage: MegaObjectStorageWrapper::new(Arc::new(CountingStorage { + inner: fixture.state.storage.git_service.obj_storage.clone(), + counts: other_counts.clone(), + })), + }; + let other_app = router(&other_state); + let request = fixture.request("GET", "chunk-map?path=/file", Body::empty()); + let rejected = tokio::spawn(async move { other_app.oneshot(request).await.unwrap() }); + wait_owners(&flight, 9).await; // Leader, seven callers and this observer. + fixture.counts.assert(1, fixture.raw.len()); + release.notify_one(); + let expected = success_json(leader.await.unwrap()).await; + for task in joined { + assert_eq!( + success_json(task.await.unwrap()).await["map"], + expected["map"] + ); + } + error(rejected.await.unwrap(), 502, "INTEGRITY_ERROR", false).await; + assert_eq!(other_counts.receipt_reads.load(Ordering::SeqCst), 1); + assert_eq!(other_counts.whole.load(Ordering::SeqCst), 0); + fixture.counts.assert(1, fixture.raw.len()); + assert_eq!(fixture.counts.receipt_writes.load(Ordering::SeqCst), 1); + assert_eq!(fixture.counts.receipt_reads.load(Ordering::SeqCst), 8); + wait_owners(&flight, 1).await; +} + +#[tokio::test] +async fn failed_or_cancelled_leader_and_cancelled_waiter_leave_actual_http_retry_capacity() { + for mode in 0..3 { + let fixture = Fixture::new().await; + let budget = MemoryBudget::new(8 * 1024 * 1024); + let repository = PostgresChunkMapRepository::new( + fixture + .state + .storage + .mono_storage() + .get_connection() + .clone(), + ) + .await + .unwrap() + .with_test_budget(budget.clone()); + assert!( + fixture + .state + .storage + .native_chunk_maps + .set(repository) + .is_ok() + ); + let (entered, release, leader) = held_leader(&fixture); + timeout(Duration::from_secs(10), entered.notified()) + .await + .unwrap(); + let flight = observer(&fixture).await; + let app = fixture.app.clone(); + let request = fixture.request("GET", "chunk-map?path=/alias", Body::empty()); + let waiter = tokio::spawn(async move { app.oneshot(request).await.unwrap() }); + wait_owners(&flight, 3).await; + assert!(budget.used() > 4 * 1024 * 1024); + *fixture.counts.receipt_write_holds.lock().unwrap() = None; + match mode { + 0 => { + fixture + .counts + .receipt_write_fail_after_create + .store(true, Ordering::SeqCst); + release.notify_one(); + error(leader.await.unwrap(), 503, "TEMPORARY_UNAVAILABLE", true).await; + success_json(waiter.await.unwrap()).await; + } + 1 => { + leader.abort(); + assert!(leader.await.err().unwrap().is_cancelled()); + success_json(waiter.await.unwrap()).await; + } + _ => { + waiter.abort(); + assert!(waiter.await.err().unwrap().is_cancelled()); + wait_owners(&flight, 2).await; + release.notify_one(); + success_json(leader.await.unwrap()).await; + } + } + wait_owners(&flight, 1).await; + drop(flight); + assert_eq!(budget.used(), 0); + let passes = if mode == 2 { 1 } else { 2 }; + fixture.counts.assert(passes, passes * fixture.raw.len()); + assert_eq!(fixture.counts.receipt_writes.load(Ordering::SeqCst), passes); + fixture.counts.reset(); + fixture.map("/file").await; + fixture.counts.assert(0, 0); + assert_eq!(budget.used(), 0); + } +} + +#[tokio::test] +async fn actual_http_long_escaped_legal_path_has_bounded_owned_json_and_empty_files_have_no_map() { + let component = "\u{1}".repeat(255); + let mut components = vec![component; 15]; + components.push("\u{1}".repeat(239)); + let name = components.join("/"); + let path = format!("/{name}"); + assert_eq!(path.len(), 4080); + crate::ceres::snapshot::view::validate_scope_relative_path(&path).unwrap(); + let fixture = Fixture::new_with_pg_config_directories_and_objects( + false, + 0, + &[(name, b"escaped path body".to_vec())], + ) + .await; + let encoded = url::form_urlencoded::byte_serialize(path.as_bytes()).collect::(); + for whole in [1, 0] { + fixture.counts.reset(); + let response = fixture + .send("GET", &format!("chunk-map?path={encoded}"), Body::empty()) + .await; + assert_eq!(response.status(), 200); + assert_eq!( + response.headers()["cache-control"], + "private, no-cache, no-transform" + ); + let bytes = to_bytes(response.into_body(), 64 * 1024).await.unwrap(); + assert!(bytes.len() > 4 * 1024); + let value: Value = serde_json::from_slice(&bytes).unwrap(); + assert_eq!(value["path"], path); + assert_eq!(value["map"]["file_size"], "17"); + fixture + .counts + .assert(whole, if whole == 0 { 0 } else { 17 }); + } + fixture.counts.reset(); + for suffix in ["chunk-map?path=/empty", "chunk-map/pages?path=/empty"] { + error( + fixture.send("GET", suffix, Body::empty()).await, + 400, + "SCOPE_INVALID", + false, + ) + .await; + } + fixture.counts.assert(0, 0); +} + +#[tokio::test] +async fn actual_http_and_repository_initialization_ignore_poisoned_temp_fact_scope_and_map_shadows() +{ + let fixture = Fixture::new_with_pg_config(true).await; + let expected = fixture.map("/file").await; + let state = reconstructed(&fixture).await; + assert!(state.storage.native_chunk_maps.get().is_none()); + let mono = state.storage.mono_storage(); + let db = mono.get_connection(); + let schema = fixture + ._schema + .as_ref() + .unwrap() + .schema() + .replace('"', "\"\""); + let tables = [ + "mst2_verified_object", + "mst2_metadata_storage_scope", + "mst2_chunk_map", + "mst2_chunk_map_source", + "mst2_chunk_map_leaf", + "mst2_chunk_map_node", + ]; + for table in tables { + db.execute_unprepared(&format!("CREATE TEMP TABLE {table}(LIKE \"{schema}\".{table}); INSERT INTO pg_temp.{table} SELECT * FROM \"{schema}\".{table}")).await.unwrap(); + } + db.execute_unprepared("UPDATE pg_temp.mst2_verified_object SET raw_sha256=decode(repeat('09',32),'hex'); UPDATE pg_temp.mst2_metadata_storage_scope SET storage_uuid='temp-poison'; UPDATE pg_temp.mst2_chunk_map SET descriptor=decode('00','hex'); UPDATE pg_temp.mst2_chunk_map_leaf SET payload=decode('00','hex'); UPDATE pg_temp.mst2_chunk_map_node SET digest=decode(repeat('09',32),'hex')").await.unwrap(); + fixture.counts.reset(); + let app = router(&state); + let map = success_json( + app.clone() + .oneshot(fixture.request("GET", "chunk-map?path=/file", Body::empty())) + .await + .unwrap(), + ) + .await; + assert_eq!(map, expected); + let map_id = map["map"]["map_id"].as_str().unwrap(); + let page = success_json( + app.clone() + .oneshot(fixture.request( + "GET", + &format!("chunk-map/pages?path=/file&map_id={map_id}&page_index=0"), + Body::empty(), + )) + .await + .unwrap(), + ) + .await; + let oracle = ChunkProjection::build(fixture.digest, fixture.raw.clone()).unwrap(); + assert_eq!( + STANDARD + .decode(page["leaf_base64"].as_str().unwrap()) + .unwrap(), + oracle.leaf_and_proof(0).unwrap().0.encode().unwrap() + ); + let body = fixture.chunk_body("/file", map_id, "1").to_string(); + let response = app + .clone() + .oneshot(fixture.request("POST", "chunks", Body::from(body))) + .await + .unwrap(); + assert_eq!(response.status(), 200); + let wire = to_bytes(response.into_body(), 2 * 1024 * 1024) + .await + .unwrap(); + let frames = parse_stream(&wire).unwrap(); + let [Frame::Chunk(chunk), Frame::End(end)] = frames.as_slice() else { + panic!("expected CHUNK and END"); + }; + assert_eq!(chunk.chunk_bytes, fixture.raw[CHUNK_SIZE as usize..]); + assert_eq!(end.logical_bytes, 113); + assert_eq!(fixture.counts.whole.load(Ordering::SeqCst), 0); + assert_eq!(fixture.counts.range.load(Ordering::SeqCst), 1); + assert_eq!(fixture.counts.bytes.load(Ordering::SeqCst), 113); + assert_eq!(fixture.counts.receipt_writes.load(Ordering::SeqCst), 0); + // Real primary changes still fail despite an apparently healthy shadow. + let actual = fixture.state.storage.mono_storage(); + let source = ChunkMapSource::from_fact(fixture.fact().await, &fixture.oid).unwrap(); + let repository = state.storage.chunk_maps().await.unwrap(); + actual.get_connection().execute_unprepared("ALTER TABLE mst2_metadata_storage_scope DISABLE TRIGGER USER; UPDATE mst2_metadata_storage_scope SET storage_uuid='changed-real-primary'; ALTER TABLE mst2_metadata_storage_scope ENABLE TRIGGER USER").await.unwrap(); + assert_eq!( + repository + .read(&source, &state.storage.git_service.obj_storage) + .await + .err() + .unwrap() + .code, + crate::ceres::snapshot::error::SnapshotErrorCode::IntegrityError + ); + error( + app.oneshot(fixture.request("GET", "chunk-map?path=/file", Body::empty())) + .await + .unwrap(), + 502, + "INTEGRITY_ERROR", + false, + ) + .await; +} + +pub(super) async fn assert_three_page_proofs_and_selected_sibling_faults( + fixture: &Fixture, + map_id: &str, + digest: [u8; 32], + pattern: &[u8], +) { + use mst2_codec::chunkmap::{ChunkMap, ProofSide, leaf_proof, merkle_root}; + let full: [u8; 32] = Sha256::digest(pattern).into(); + let final_chunk: [u8; 32] = Sha256::digest(&pattern[..7]).into(); + let leaves = [ + ChunkLeaf { + page_index: 0, + chunk_sha256: vec![full; 256], + }, + ChunkLeaf { + page_index: 1, + chunk_sha256: vec![full; 256], + }, + ChunkLeaf { + page_index: 2, + chunk_sha256: vec![final_chunk], + }, + ]; + let hashes: Vec<_> = leaves + .iter() + .map(|leaf| leaf.leaf_hash().unwrap()) + .collect(); + let root = merkle_root(&hashes).unwrap(); + let oracle = ChunkMap::new(digest, 512 * CHUNK_SIZE as u64 + 7, root).unwrap(); + assert_eq!(map_id, format!("sha256:{}", hex_of(&oracle.map_id()))); + for index in 0..3 { + let page = success_json( + fixture + .send( + "GET", + &format!("chunk-map/pages?path=/file&map_id={map_id}&page_index={index}"), + Body::empty(), + ) + .await, + ) + .await; + assert_eq!( + STANDARD + .decode(page["leaf_base64"].as_str().unwrap()) + .unwrap(), + leaves[index as usize].encode().unwrap() + ); + let proof = leaf_proof(&hashes, index).unwrap(); + let expected: Vec<_> = proof.iter().map(|step| json!({ + "side": if step.side == ProofSide::Left { "left" } else { "right" }, + "sibling_pages": step.sibling_pages.to_string(), "digest": format!("sha256:{}", hex_of(&step.digest)), + })).collect(); + assert_eq!(page["proof"], json!(expected)); + verify_leaf( + 3, + index, + leaves[index as usize].leaf_hash().unwrap(), + &proof, + root, + ) + .unwrap(); + } + fixture.counts.assert(0, 0); + let mono = fixture.state.storage.mono_storage(); + let db = mono.get_connection(); + let id = oracle.map_id().to_vec(); + for missing in [false, true] { + db.execute_unprepared("ALTER TABLE mst2_chunk_map_node DISABLE TRIGGER USER") + .await + .unwrap(); + if missing { + db.execute_raw(statement( + "DELETE FROM mst2_chunk_map_node WHERE map_id=$1 AND first_page=2 AND page_count=1", + [id.clone().into()], + )) + .await + .unwrap(); + } else { + db.execute_raw(statement("UPDATE mst2_chunk_map_node SET digest=$2 WHERE map_id=$1 AND first_page=2 AND page_count=1", [id.clone().into(), vec![9; 32].into()])).await.unwrap(); + } + db.execute_unprepared("ALTER TABLE mst2_chunk_map_node ENABLE TRIGGER USER") + .await + .unwrap(); + for (method, suffix, body) in [ + ( + "GET", + format!("chunk-map/pages?path=/file&map_id={map_id}&page_index=0"), + Body::empty(), + ), + ( + "POST", + "chunks".to_string(), + Body::from(fixture.chunk_body("/file", map_id, "0").to_string()), + ), + ] { + error( + fixture.send(method, &suffix, body).await, + 502, + "INTEGRITY_ERROR", + false, + ) + .await; + } + fixture.counts.assert(0, 0); + // Page 2 proves itself through [0,2); the unrelated damaged node is + // absent from its selected SQL and does not turn into a full-map scan. + let page = success_json( + fixture + .send( + "GET", + &format!("chunk-map/pages?path=/file&map_id={map_id}&page_index=2"), + Body::empty(), + ) + .await, + ) + .await; + assert_eq!( + STANDARD + .decode(page["leaf_base64"].as_str().unwrap()) + .unwrap(), + leaves[2].encode().unwrap() + ); + db.execute_unprepared("ALTER TABLE mst2_chunk_map_node DISABLE TRIGGER USER") + .await + .unwrap(); + if missing { + db.execute_raw(statement("INSERT INTO mst2_chunk_map_node(map_id,first_page,page_count,digest) VALUES($1,2,1,$2)", [id.clone().into(), hashes[2].to_vec().into()])).await.unwrap(); + } else { + db.execute_raw(statement("UPDATE mst2_chunk_map_node SET digest=$2 WHERE map_id=$1 AND first_page=2 AND page_count=1", [id.clone().into(), hashes[2].to_vec().into()])).await.unwrap(); + } + db.execute_unprepared("ALTER TABLE mst2_chunk_map_node ENABLE TRIGGER USER") + .await + .unwrap(); + } + fixture.counts.assert(0, 0); +} + +#[tokio::test] +async fn failed_digest_failed_body_and_cancelled_cold_producer_allow_the_joined_current_source_to_retry() + { + use super::bounded_objects::{FaultKind, StreamFault}; + for mode in 0..3 { + let fixture = Fixture::new().await; + let budget = MemoryBudget::new(8 * 1024 * 1024); + let repository = PostgresChunkMapRepository::new( + fixture + .state + .storage + .mono_storage() + .get_connection() + .clone(), + ) + .await + .unwrap() + .with_test_budget(budget.clone()); + assert!( + fixture + .state + .storage + .native_chunk_maps + .set(repository) + .is_ok() + ); + let entered = Arc::new(Notify::new()); + let release = Arc::new(Notify::new()); + let drops = Arc::new(AtomicUsize::new(0)); + let kind = if mode == 1 { + FaultKind::HeldThenError { + entered: entered.clone(), + release: release.clone(), + drops: drops.clone(), + } + } else { + let mut raw = fixture.raw.clone(); + if mode == 0 { + raw[0] ^= 1; + } + FaultKind::Held { + raw: Bytes::from(raw), + entered: entered.clone(), + release: release.clone(), + drops: drops.clone(), + } + }; + *fixture.counts.object_fault.lock().unwrap() = Some(StreamFault { + oid: fixture.oid.clone(), + kind, + }); + let app = fixture.app.clone(); + let request = fixture.request("GET", "chunk-map?path=/file", Body::empty()); + let leader = tokio::spawn(async move { app.oneshot(request).await.unwrap() }); + timeout(Duration::from_secs(10), entered.notified()) + .await + .unwrap(); + let flight = observer(&fixture).await; + let app = fixture.app.clone(); + let request = fixture.request("GET", "chunk-map?path=/alias", Body::empty()); + let waiter = tokio::spawn(async move { app.oneshot(request).await.unwrap() }); + wait_owners(&flight, 3).await; + assert_eq!(fixture.counts.receipt_writes.load(Ordering::SeqCst), 0); + assert!(budget.used() > 4 * 1024 * 1024); + *fixture.counts.object_fault.lock().unwrap() = None; + if mode == 2 { + leader.abort(); + assert!(leader.await.err().unwrap().is_cancelled()); + } else { + release.notify_one(); + error( + leader.await.unwrap(), + if mode == 0 { 502 } else { 503 }, + if mode == 0 { + "INTEGRITY_ERROR" + } else { + "OBJECT_UNAVAILABLE" + }, + false, + ) + .await; + } + let map = success_json(waiter.await.unwrap()).await; + assert_eq!( + map["map"]["file_content_id"], + format!("sha256:{}", hex_of(&fixture.digest)) + ); + assert_eq!(drops.load(Ordering::SeqCst), 1); + assert_eq!(fixture.counts.whole.load(Ordering::SeqCst), 2); + assert_eq!(fixture.counts.receipt_writes.load(Ordering::SeqCst), 1); + let failed_bytes = match mode { + 0 => fixture.raw.len(), + 1 => 0, + _ => 1, + }; + assert_eq!( + fixture.counts.bytes.load(Ordering::SeqCst), + fixture.raw.len() + failed_bytes + ); + wait_owners(&flight, 1).await; + drop(flight); + assert_eq!(budget.used(), 0); + fixture.counts.reset(); + fixture.map("/file").await; + fixture.counts.assert(0, 0); + } +} + +#[tokio::test] +async fn different_current_sources_enter_cold_installations_independently() { + let fixture = Fixture::new_with_pg_config_directories_and_objects( + false, + 0, + &[("other".into(), b"independent source".to_vec())], + ) + .await; + let (entered, release, leader) = held_leader(&fixture); + timeout(Duration::from_secs(10), entered.notified()) + .await + .unwrap(); + let app = fixture.app.clone(); + let request = fixture.request("GET", "chunk-map?path=/other", Body::empty()); + let other = tokio::spawn(async move { app.oneshot(request).await.unwrap() }); + timeout(Duration::from_secs(10), entered.notified()) + .await + .unwrap(); + assert_eq!(fixture.counts.whole.load(Ordering::SeqCst), 2); + assert_eq!(fixture.counts.receipt_writes.load(Ordering::SeqCst), 2); + release.notify_waiters(); + let map = success_json(leader.await.unwrap()).await; + let second = success_json(other.await.unwrap()).await; + assert_ne!(map["map"]["map_id"], second["map"]["map_id"]); + fixture.counts.assert(2, fixture.raw.len() + 18); +} + +#[tokio::test] +async fn persisted_descriptors_and_authenticated_pages_charge_until_the_last_live_reader_drops() { + let fixture = Fixture::new().await; + fixture.map("/file").await; + fixture.counts.reset(); + let budget = MemoryBudget::new(96 * 1024); + let repository = PostgresChunkMapRepository::new( + fixture + .state + .storage + .mono_storage() + .get_connection() + .clone(), + ) + .await + .unwrap() + .with_test_budget(budget.clone()); + let source = ChunkMapSource::from_fact(fixture.fact().await, &fixture.oid).unwrap(); + let map = repository + .read(&source, &fixture.state.storage.git_service.obj_storage) + .await + .unwrap() + .unwrap(); + let page = repository.selected_page(&map, 0).await.unwrap(); + let other_map = map.clone(); + let other_page = page.clone(); + drop(map); + drop(page); + assert_eq!(budget.used(), 96 * 1024); + assert_eq!( + budget.reserve(1).err().unwrap().code, + SnapshotErrorCode::TemporaryUnavailable + ); + other_page + .verify_chunk(&other_map.map, 1, &fixture.raw[CHUNK_SIZE as usize..]) + .unwrap(); + drop(other_map); + assert_eq!(budget.used(), 64 * 1024); + drop(other_page); + assert_eq!(budget.used(), 0); + let replay = repository + .read(&source, &fixture.state.storage.git_service.obj_storage) + .await + .unwrap() + .unwrap(); + assert_eq!(budget.used(), 32 * 1024); + drop(replay); + assert_eq!(budget.used(), 0); + fixture.counts.assert(0, 0); +} + +#[tokio::test] +async fn actual_install_sql_failure_rolls_back_every_index_and_exact_retry_earns_admission_again() { + let fixture = Fixture::new().await; + let mono = fixture.state.storage.mono_storage(); + let db = mono.get_connection(); + let budget = MemoryBudget::new(8 * 1024 * 1024); + let repository = PostgresChunkMapRepository::new(db.clone()) + .await + .unwrap() + .with_test_budget(budget.clone()); + assert!( + fixture + .state + .storage + .native_chunk_maps + .set(repository) + .is_ok() + ); + db.execute_unprepared("CREATE FUNCTION chunk_map_install_test_failure() RETURNS trigger LANGUAGE plpgsql AS $test$ BEGIN RAISE EXCEPTION 'injected source-row failure after complete index insertion'; END $test$; CREATE TRIGGER chunk_map_install_test_failure BEFORE INSERT ON mst2_chunk_map_source FOR EACH ROW EXECUTE FUNCTION chunk_map_install_test_failure()").await.unwrap(); + error( + fixture + .send("GET", "chunk-map?path=/file", Body::empty()) + .await, + 503, + "TEMPORARY_UNAVAILABLE", + true, + ) + .await; + fixture.counts.assert(1, fixture.raw.len()); + assert_eq!(fixture.counts.receipt_writes.load(Ordering::SeqCst), 1); + assert_eq!(budget.used(), 0); + for table in [ + "mst2_chunk_map", + "mst2_chunk_map_leaf", + "mst2_chunk_map_node", + "mst2_chunk_map_source", + ] { + assert_eq!(count(db, table).await, 0); + } + db.execute_unprepared("DROP TRIGGER chunk_map_install_test_failure ON mst2_chunk_map_source; DROP FUNCTION chunk_map_install_test_failure()").await.unwrap(); + fixture.counts.reset(); + let map = fixture.map("/file").await; + fixture.counts.assert(1, fixture.raw.len()); + assert_eq!(fixture.counts.receipt_writes.load(Ordering::SeqCst), 1); + assert_eq!(budget.used(), 0); + for table in [ + "mst2_chunk_map", + "mst2_chunk_map_leaf", + "mst2_chunk_map_node", + "mst2_chunk_map_source", + ] { + assert_eq!(count(db, table).await, 1); + } + fixture.counts.reset(); + assert_eq!(fixture.map("/alias").await["map"], map["map"]); + fixture.counts.assert(0, 0); + assert_eq!(fixture.counts.receipt_writes.load(Ordering::SeqCst), 0); +} + +#[tokio::test] +async fn ordinary_incomplete_source_dml_cannot_commit_an_admitted_partial_map() { + let fixture = Fixture::new().await; + let mono = fixture.state.storage.mono_storage(); + let db = mono.get_connection(); + let source = ChunkMapSource::from_fact(fixture.fact().await, &fixture.oid).unwrap(); + let repository = PostgresChunkMapRepository::new(db.clone()).await.unwrap(); + let map = ChunkProjection::build(fixture.digest, fixture.raw.clone()).unwrap(); + let scope = repository.test_primary_scope(); + let txn = db + .begin_with_config(Some(IsolationLevel::ReadCommitted), None) + .await + .unwrap(); + txn.execute_raw(statement( + "INSERT INTO mst2_chunk_map(map_id,descriptor,page_count,pages_root) VALUES($1,$2,$3,$4)", + [ + map.map_id.to_vec().into(), + map.map.encode().into(), + (map.map.page_count as i32).into(), + map.map.pages_root.to_vec().into(), + ], + )) + .await + .unwrap(); + txn.execute_raw(statement("INSERT INTO mst2_chunk_map_source(storage_domain,git_oid,object_kind,fact_id,source_id,source_bytes,primary_scope,map_id,receipt_digest) VALUES('git',$1,'blob',$2,$3,$4,$5,$6,$7)", [fixture.oid.clone().into(), source.fact().id.into(), repository.source_identity(&source).unwrap().to_vec().into(), source.canonical_bytes().unwrap().into(), scope.to_vec().into(), map.map_id.to_vec().into(), vec![9;32].into()])).await.unwrap(); + assert!(txn.commit().await.is_err()); + for table in [ + "mst2_chunk_map", + "mst2_chunk_map_leaf", + "mst2_chunk_map_node", + "mst2_chunk_map_source", + ] { + assert_eq!(count(db, table).await, 0); + } + fixture.counts.assert(0, 0); +} + +#[tokio::test] +async fn actual_map_and_page_json_revalidate_lease_before_the_first_transport_poll() { + for page in [false, true] { + let fixture = Fixture::new().await; + let map = fixture.map("/file").await; + let suffix = if page { + format!( + "chunk-map/pages?path=/file&map_id={}&page_index=0", + map["map"]["map_id"].as_str().unwrap() + ) + } else { + "chunk-map?path=/file".into() + }; + fixture.counts.reset(); + let response = fixture.send("GET", &suffix, Body::empty()).await; + assert_eq!(response.status(), 200); + let revoked = fixture + .app + .clone() + .oneshot( + Request::builder() + .method("DELETE") + .uri(format!("/api/v2/snapshots/leases/{}", fixture.lease)) + .header("authorization", format!("Bearer {TOKEN}")) + .body(Body::empty()) + .unwrap(), + ) + .await + .unwrap(); + assert_eq!(revoked.status(), 200); + let mut stream = response.into_body().into_data_stream(); + assert!(stream.next().await.unwrap().is_err()); + assert!(stream.next().await.is_none()); + fixture.counts.assert(0, 0); + } +} + +#[tokio::test] +async fn oversized_current_fact_digest_fails_before_source_or_receipt_io_and_recovers_canonical_bytes() + { + let fixture = Fixture::new().await; + let expected = fixture.map("/file").await; + let original = fixture.fact().await; + let mono = fixture.state.storage.mono_storage(); + mono.get_connection().execute_raw(statement("UPDATE mst2_verified_object SET raw_sha256=decode(repeat('ab',1048576),'hex') WHERE id=$1", [original.id.into()])).await.unwrap(); + fixture.counts.reset(); + error( + fixture + .send("GET", "chunk-map?path=/file", Body::empty()) + .await, + 502, + "INTEGRITY_ERROR", + false, + ) + .await; + fixture.counts.assert(0, 0); + assert_eq!(fixture.counts.receipt_reads.load(Ordering::SeqCst), 0); + fixture.replace_fact(original).await; + assert_eq!(fixture.map("/file").await, expected); + fixture.counts.assert(0, 0); +} + +#[tokio::test] +async fn warm_admission_requires_receipt_exact_bytes_size_and_final_eof_without_fallback() { + let fixture = Fixture::new().await; + let map = fixture.map("/file").await; + let repository = fixture.state.storage.chunk_maps().await.unwrap(); + let source = ChunkMapSource::from_fact(fixture.fact().await, &fixture.oid).unwrap(); + let key = ObjectKey { + namespace: ObjectNamespace::ChunkMapReceipt, + key: hex_of(&repository.source_identity(&source).unwrap()), + }; + let (mut stream, meta) = fixture + .state + .storage + .git_service + .obj_storage + .inner + .get_stream(&key) + .await + .unwrap(); + let mut original = Vec::new(); + while let Some(part) = stream.next().await { + original.extend_from_slice(&part.unwrap()); + } + assert_eq!(original.len() as i64, meta.size); + let mut wrong = original.clone(); + wrong[0] ^= 1; + let mut long = original.clone(); + long.push(0); + let cases = [ + original[..original.len() - 1].to_vec(), + long, + wrong, + original.clone(), + ]; + fixture + .counts + .receipt_read_meta_size + .store(meta.size, Ordering::SeqCst); + for (index, bytes) in cases.into_iter().enumerate() { + fixture.counts.reset(); + *fixture.counts.receipt_read_corruption.lock().unwrap() = Some(Bytes::from(bytes)); + fixture + .counts + .receipt_read_late_error + .store(index == 3, Ordering::SeqCst); + error( + fixture + .send("GET", "chunk-map?path=/file", Body::empty()) + .await, + 502, + "INTEGRITY_ERROR", + false, + ) + .await; + fixture.counts.assert(0, 0); + assert_eq!(fixture.counts.receipt_reads.load(Ordering::SeqCst), 1); + assert_eq!(fixture.counts.receipt_writes.load(Ordering::SeqCst), 0); + } + *fixture.counts.receipt_read_corruption.lock().unwrap() = None; + fixture + .counts + .receipt_read_late_error + .store(false, Ordering::SeqCst); + assert_eq!(fixture.map("/file").await, map); + fixture.counts.assert(0, 0); +} diff --git a/src/ceres/snapshot/chunk_map_gate.rs b/src/ceres/snapshot/chunk_map_gate.rs new file mode 100644 index 00000000..5c56a504 --- /dev/null +++ b/src/ceres/snapshot/chunk_map_gate.rs @@ -0,0 +1,185 @@ +//! Bounded exact-source install flights. Completed data is always re-read +//! from the requesting source's trusted receipt, never retained in a flight. + +use std::{ + collections::HashMap, + sync::{Arc, Mutex, OnceLock, Weak}, +}; + +use super::error::{SnapshotError, SnapshotErrorCode}; + +const MAX_INSTALL_FLIGHTS: usize = 128; +type SourceGate = tokio::sync::Mutex<()>; + +#[derive(Default)] +struct Registry { + entries: HashMap<[u8; 32], Weak>, +} + +pub(crate) struct InstallFlight { + key: [u8; 32], + gate: Option>, + registry: Arc>, +} + +impl InstallFlight { + pub(crate) fn acquire(key: [u8; 32]) -> Result { + static REGISTRY: OnceLock>> = OnceLock::new(); + Self::from_registry(REGISTRY.get_or_init(|| Arc::default()).clone(), key) + } + + fn from_registry(registry: Arc>, key: [u8; 32]) -> Result { + let gate = { + let mut state = registry.lock().map_err(|_| internal())?; + if let Some(gate) = state.entries.get(&key).and_then(Weak::upgrade) { + gate + } else { + state.entries.remove(&key); + if state.entries.len() >= MAX_INSTALL_FLIGHTS { + return Err(SnapshotError::new( + SnapshotErrorCode::LimitExceeded, + "too many distinct source map installations in flight", + )); + } + let gate = Arc::new(SourceGate::new(())); + state.entries.insert(key, Arc::downgrade(&gate)); + gate + } + }; + Ok(Self { + key, + gate: Some(gate), + registry, + }) + } + + pub(crate) async fn lock(&self) -> Result, SnapshotError> { + Ok(self.gate.as_ref().ok_or_else(internal)?.lock().await) + } + + #[cfg(test)] + pub(crate) fn test_owner_count(&self) -> usize { + self.gate.as_ref().map_or(0, Arc::strong_count) + } +} + +impl Drop for InstallFlight { + fn drop(&mut self) { + let Some(gate) = self.gate.take() else { + return; + }; + if let Ok(mut state) = self.registry.lock() { + if Arc::strong_count(&gate) == 1 + && state + .entries + .get(&self.key) + .is_some_and(|entry| entry.ptr_eq(&Arc::downgrade(&gate))) + { + state.entries.remove(&self.key); + } + // Owners must release their strong reference while the registry + // is locked, so concurrent final drops cannot leave a stale slot. + drop(gate); + } + } +} + +fn internal() -> SnapshotError { + SnapshotError::new( + SnapshotErrorCode::Internal, + "source map flight registry is unavailable", + ) +} + +#[cfg(test)] +mod tests { + use super::*; + + #[tokio::test] + async fn exact_source_gate_blocks_only_same_source_and_returns_capacity_after_last_owner() { + let registry = Arc::new(Mutex::new(Registry::default())); + let first = InstallFlight::from_registry(registry.clone(), [1; 32]).unwrap(); + let lock = first.lock().await.unwrap(); + let same = InstallFlight::from_registry(registry.clone(), [1; 32]).unwrap(); + assert!(same.gate.as_ref().unwrap().try_lock().is_err()); + let other = InstallFlight::from_registry(registry.clone(), [2; 32]).unwrap(); + assert!(other.gate.as_ref().unwrap().try_lock().is_ok()); + drop(lock); + drop(first); + assert!(same.gate.as_ref().unwrap().try_lock().is_ok()); + assert_eq!(registry.lock().unwrap().entries.len(), 2); + drop(same); + drop(other); + assert!(registry.lock().unwrap().entries.is_empty()); + let mut held = Vec::new(); + for i in 0..MAX_INSTALL_FLIGHTS { + let mut key = [0; 32]; + key[..8].copy_from_slice(&(i as u64).to_le_bytes()); + held.push(InstallFlight::from_registry(registry.clone(), key).unwrap()); + } + assert_eq!( + InstallFlight::from_registry(registry.clone(), [255; 32]) + .err() + .unwrap() + .code, + SnapshotErrorCode::LimitExceeded + ); + let joined_at_capacity = InstallFlight::from_registry(registry.clone(), [0; 32]).unwrap(); + assert_eq!(joined_at_capacity.test_owner_count(), 2); + assert_eq!(registry.lock().unwrap().entries.len(), MAX_INSTALL_FLIGHTS); + drop(joined_at_capacity); + drop(held); + assert!(registry.lock().unwrap().entries.is_empty()); + assert!(InstallFlight::from_registry(registry, [255; 32]).is_ok()); + } + + #[test] + fn concurrent_last_owners_release_registry_capacity() { + for _ in 0..64 { + let registry = Arc::new(Mutex::new(Registry::default())); + let first = InstallFlight::from_registry(registry.clone(), [1; 32]).unwrap(); + let second = InstallFlight::from_registry(registry.clone(), [1; 32]).unwrap(); + let barrier = Arc::new(std::sync::Barrier::new(2)); + std::thread::scope(|scope| { + let other = barrier.clone(); + scope.spawn(move || { + other.wait(); + drop(first); + }); + scope.spawn(move || { + barrier.wait(); + drop(second); + }); + }); + assert!(registry.lock().unwrap().entries.is_empty()); + } + } + + #[test] + fn dropping_an_old_claim_does_not_remove_a_replacement_gate() { + let registry = Arc::new(Mutex::new(Registry::default())); + let old = InstallFlight::from_registry(registry.clone(), [1; 32]).unwrap(); + let replacement = Arc::new(SourceGate::new(())); + registry + .lock() + .unwrap() + .entries + .insert([1; 32], Arc::downgrade(&replacement)); + drop(old); + assert_eq!(registry.lock().unwrap().entries.len(), 1); + let current = InstallFlight::from_registry(registry.clone(), [1; 32]).unwrap(); + assert!(Arc::ptr_eq(current.gate.as_ref().unwrap(), &replacement)); + drop(replacement); + drop(current); + assert!(registry.lock().unwrap().entries.is_empty()); + registry + .lock() + .unwrap() + .entries + .insert([1; 32], Weak::new()); + let reclaimed = InstallFlight::from_registry(registry.clone(), [1; 32]).unwrap(); + assert_eq!(registry.lock().unwrap().entries.len(), 1); + drop(reclaimed); + assert!(registry.lock().unwrap().entries.is_empty()); + } +} diff --git a/src/ceres/snapshot/chunk_map_index.rs b/src/ceres/snapshot/chunk_map_index.rs new file mode 100644 index 00000000..7e253f9c --- /dev/null +++ b/src/ceres/snapshot/chunk_map_index.rs @@ -0,0 +1,171 @@ +//! Canonical MCL2 Merkle subtrees addressed by their exact leaf interval. + +use mst2_codec::chunkmap::{ProofSide, ProofStep}; +use sha2::{Digest, Sha256}; + +use super::error::{SnapshotError, SnapshotErrorCode}; + +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub(crate) struct ChunkMapNode { + pub start: u64, + pub pages: u64, + pub digest: [u8; 32], +} + +pub(crate) fn indexed_nodes(hashes: &[[u8; 32]]) -> Result, SnapshotError> { + if hashes.is_empty() || hashes.len() > 32_768 { + return Err(integrity("chunk map is outside the indexed page profile")); + } + let mut nodes = Vec::new(); + nodes.try_reserve_exact(hashes.len() * 2 - 1).map_err(|_| { + SnapshotError::new( + SnapshotErrorCode::TemporaryUnavailable, + "chunk map index allocation failed", + ) + })?; + build(hashes, 0, &mut nodes); + Ok(nodes) +} + +fn build(hashes: &[[u8; 32]], start: u64, nodes: &mut Vec) -> [u8; 32] { + let pages = hashes.len() as u64; + let digest = if pages == 1 { + hashes[0] + } else { + let split = split(pages); + let left = build(&hashes[..split as usize], start, nodes); + let right = build(&hashes[split as usize..], start + split, nodes); + let mut hash = Sha256::new(); + hash.update(b"mega.mst2.chunkbranch\0"); + hash.update(split.to_le_bytes()); + hash.update(left); + hash.update((pages - split).to_le_bytes()); + hash.update(right); + hash.finalize().into() + }; + nodes.push(ChunkMapNode { + start, + pages, + digest, + }); + digest +} + +fn split(pages: u64) -> u64 { + 1 << (63 - (pages - 1).leading_zeros()) +} + +/// Only these sibling intervals may be fetched for a selected page. +pub(crate) fn proof_intervals( + pages: u64, + index: u64, +) -> Result, SnapshotError> { + if pages == 0 || pages > 32_768 || index >= pages { + return Err(SnapshotError::new( + SnapshotErrorCode::PathNotFound, + "chunk map page does not exist", + )); + } + let (mut start, mut count) = (0, pages); + let mut intervals = Vec::new(); + while count > 1 { + let left = split(count); + if index < start + left { + intervals.push((ProofSide::Right, start + left, count - left)); + count = left; + } else { + intervals.push((ProofSide::Left, start, left)); + start += left; + count -= left; + } + } + intervals.reverse(); + Ok(intervals) +} + +pub(crate) fn selected_proof( + intervals: &[(ProofSide, u64, u64)], + nodes: &[ChunkMapNode], +) -> Result, SnapshotError> { + if nodes.len() != intervals.len() { + return Err(integrity( + "persisted chunk map proof coverage is incomplete", + )); + } + intervals + .iter() + .map(|&(side, start, pages)| { + let mut matching = nodes + .iter() + .filter(|n| n.start == start && n.pages == pages); + let node = matching + .next() + .ok_or_else(|| integrity("persisted chunk map sibling is missing"))?; + if matching.next().is_some() { + return Err(integrity("persisted chunk map sibling is duplicated")); + } + Ok(ProofStep { + side, + sibling_pages: pages, + digest: node.digest, + }) + }) + .collect() +} + +fn integrity(message: &str) -> SnapshotError { + SnapshotError::new(SnapshotErrorCode::IntegrityError, message) +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn indexed_selected_proofs_match_independent_codec_for_uneven_trees() { + for count in [1, 2, 3, 5, 9, 17, 257, 32_768] { + let hashes: Vec<[u8; 32]> = (0u64..count) + .map(|n| Sha256::digest(n.to_le_bytes()).into()) + .collect(); + let nodes = indexed_nodes(&hashes).unwrap(); + assert_eq!(nodes.len(), hashes.len() * 2 - 1); + let root = mst2_codec::chunkmap::merkle_root(&hashes).unwrap(); + assert_eq!(nodes.last().unwrap().digest, root); + for page in [0, count / 2, count - 1] { + let intervals = proof_intervals(count, page).unwrap(); + assert!(intervals.len() <= 15); + let selected: Vec<_> = nodes + .iter() + .filter(|n| { + intervals + .iter() + .any(|&(_, s, c)| n.start == s && n.pages == c) + }) + .copied() + .collect(); + let proof = selected_proof(&intervals, &selected).unwrap(); + assert_eq!( + proof, + mst2_codec::chunkmap::leaf_proof(&hashes, page).unwrap() + ); + mst2_codec::chunkmap::verify_leaf(count, page, hashes[page as usize], &proof, root) + .unwrap(); + if !selected.is_empty() { + assert!(selected_proof(&intervals, &selected[1..]).is_err()); + let mut bad = proof.clone(); + bad[0].digest[0] ^= 1; + assert!( + mst2_codec::chunkmap::verify_leaf( + count, + page, + hashes[page as usize], + &bad, + root + ) + .is_err() + ); + } + } + } + } +} diff --git a/src/ceres/snapshot/chunk_singleflight_tests.rs b/src/ceres/snapshot/chunk_singleflight_tests.rs deleted file mode 100644 index 78daff83..00000000 --- a/src/ceres/snapshot/chunk_singleflight_tests.rs +++ /dev/null @@ -1,487 +0,0 @@ -use std::{ - sync::{ - Barrier, - atomic::{AtomicUsize, Ordering}, - }, - time::Duration, -}; - -use tokio::{sync::Notify, time::timeout}; - -use super::*; - -fn unique_data(size: usize) -> ([u8; 32], Arc>) { - let seed = uuid::Uuid::new_v4(); - let raw: Vec = (0..size).map(|i| seed.as_bytes()[i % 16]).collect(); - (Sha256::digest(&raw).into(), Arc::new(raw)) -} - -async fn started(signal: &Notify) { - timeout(Duration::from_secs(5), signal.notified()) - .await - .unwrap(); -} - -async fn participants(content_id: [u8; 32], expected: usize) { - timeout(Duration::from_secs(5), async { - loop { - let count = FLIGHTS - .get() - .unwrap() - .lock() - .unwrap() - .entries - .get(&content_id) - .map_or(0, Weak::strong_count); - if count == expected { - break; - } - tokio::task::yield_now().await; - } - }) - .await - .unwrap(); -} - -fn assert_flight_released(content_id: [u8; 32]) { - assert!( - !FLIGHTS - .get() - .unwrap() - .lock() - .unwrap() - .entries - .contains_key(&content_id) - ); -} - -fn assert_uncached(content_id: [u8; 32]) { - assert!( - STAGED - .get() - .unwrap() - .lock() - .unwrap() - .get(content_id) - .is_none() - ); -} - -fn evict(content_id: [u8; 32]) { - let mut cache = STAGED.get().unwrap().lock().unwrap(); - let projection = cache.entries.remove(&content_id).unwrap(); - cache.total_bytes -= projection.retained_bytes(); - let position = cache.order.iter().position(|id| *id == content_id).unwrap(); - cache.order.remove(position); -} - -#[tokio::test] -async fn cold_same_digest_loads_once_and_warm_cache_skips_loader() { - let (content_id, raw) = unique_data(CHUNK_SIZE as usize + 7); - let calls = Arc::new(AtomicUsize::new(0)); - let entered = Arc::new(Notify::new()); - let release = Arc::new(Notify::new()); - let leader = tokio::spawn({ - let raw = raw.clone(); - let calls = calls.clone(); - let entered = entered.clone(); - let release = release.clone(); - async move { - get_or_project(content_id, || async move { - calls.fetch_add(1, Ordering::SeqCst); - entered.notify_one(); - release.notified().await; - Ok(raw.as_ref().clone()) - }) - .await - } - }); - started(&entered).await; - let mut waiters = Vec::new(); - for _ in 0..7 { - let raw = raw.clone(); - let calls = calls.clone(); - waiters.push(tokio::spawn(async move { - get_or_project(content_id, || async move { - calls.fetch_add(1, Ordering::SeqCst); - Ok(raw.as_ref().clone()) - }) - .await - })); - } - participants(content_id, 8).await; - release.notify_one(); - let projection = leader.await.unwrap().unwrap(); - for waiter in waiters { - let shared = waiter.await.unwrap().unwrap(); - assert!(Arc::ptr_eq(&projection, &shared)); - } - assert_eq!(calls.load(Ordering::SeqCst), 1); - assert_eq!(projection.map.file_content_id, content_id); - assert_eq!(projection.map.chunk_count, 2); - assert_eq!( - projection.chunk_bytes(0).unwrap(), - &raw[..CHUNK_SIZE as usize] - ); - assert_eq!( - projection.chunk_bytes(1).unwrap(), - &raw[CHUNK_SIZE as usize..] - ); - let (leaf, proof) = projection.leaf_and_proof(0).unwrap(); - mst2_codec::chunkmap::verify_leaf( - projection.page_count(), - 0, - leaf.leaf_hash().unwrap(), - &proof, - projection.map.pages_root, - ) - .unwrap(); - let warm = get_or_project(content_id, || async { - panic!("warm projection must not invoke its loader"); - }) - .await - .unwrap(); - assert!(Arc::ptr_eq(&projection, &warm)); - assert_flight_released(content_id); -} - -#[tokio::test] -async fn different_digests_enter_their_loaders_independently() { - let mut requests = Vec::new(); - let mut signals = Vec::new(); - for _ in 0..2 { - let (content_id, raw) = unique_data(97); - let entered = Arc::new(Notify::new()); - let release = Arc::new(Notify::new()); - signals.push((content_id, entered.clone(), release.clone())); - requests.push(tokio::spawn(async move { - get_or_project(content_id, || async move { - entered.notify_one(); - release.notified().await; - Ok(raw.as_ref().clone()) - }) - .await - })); - } - for (_, entered, _) in &signals { - started(entered).await; - } - for (_, _, release) in &signals { - release.notify_one(); - } - for request in requests { - request.await.unwrap().unwrap(); - } - for (content_id, _, _) in signals { - assert_flight_released(content_id); - } -} - -async fn failed_leader_then_waiter(wrong_digest: bool) { - let (content_id, raw) = unique_data(103); - let calls = Arc::new(AtomicUsize::new(0)); - let first_entered = Arc::new(Notify::new()); - let first_release = Arc::new(Notify::new()); - let second_entered = Arc::new(Notify::new()); - let second_release = Arc::new(Notify::new()); - let leader = tokio::spawn({ - let calls = calls.clone(); - let entered = first_entered.clone(); - let release = first_release.clone(); - async move { - get_or_project(content_id, || async move { - calls.fetch_add(1, Ordering::SeqCst); - entered.notify_one(); - release.notified().await; - if wrong_digest { - Ok(vec![0; 103]) - } else { - Err(SnapshotError::new( - SnapshotErrorCode::ObjectUnavailable, - "failed test loader", - )) - } - }) - .await - } - }); - started(&first_entered).await; - let waiter = tokio::spawn({ - let raw = raw.clone(); - let calls = calls.clone(); - let entered = second_entered.clone(); - let release = second_release.clone(); - async move { - get_or_project(content_id, || async move { - calls.fetch_add(1, Ordering::SeqCst); - entered.notify_one(); - release.notified().await; - Ok(raw.as_ref().clone()) - }) - .await - } - }); - participants(content_id, 2).await; - first_release.notify_one(); - let failure = leader.await.unwrap().err().unwrap(); - assert_eq!( - failure.code, - if wrong_digest { - SnapshotErrorCode::DigestMismatch - } else { - SnapshotErrorCode::ObjectUnavailable - } - ); - started(&second_entered).await; - assert_eq!(calls.load(Ordering::SeqCst), 2); - assert_uncached(content_id); - second_release.notify_one(); - let projection = waiter.await.unwrap().unwrap(); - assert_eq!(projection.chunk_bytes(0).unwrap(), raw.as_slice()); - assert_flight_released(content_id); -} - -#[tokio::test] -async fn load_failure_is_unpublished_and_waiter_uses_its_loader() { - failed_leader_then_waiter(false).await; -} - -#[tokio::test] -async fn digest_failure_is_unpublished_and_waiter_uses_its_loader() { - failed_leader_then_waiter(true).await; -} - -#[tokio::test] -async fn cancelled_leader_releases_gate_for_waiting_loader() { - let (content_id, raw) = unique_data(113); - let entered = Arc::new(Notify::new()); - let release = Arc::new(Notify::new()); - let takeover = Arc::new(Notify::new()); - let takeover_release = Arc::new(Notify::new()); - let calls = Arc::new(AtomicUsize::new(0)); - let leader = tokio::spawn({ - let raw = raw.clone(); - let calls = calls.clone(); - let entered = entered.clone(); - let release = release.clone(); - async move { - get_or_project(content_id, || async move { - calls.fetch_add(1, Ordering::SeqCst); - entered.notify_one(); - release.notified().await; - Ok(raw.as_ref().clone()) - }) - .await - } - }); - started(&entered).await; - let waiter = tokio::spawn({ - let raw = raw.clone(); - let calls = calls.clone(); - let takeover = takeover.clone(); - let takeover_release = takeover_release.clone(); - async move { - get_or_project(content_id, || async move { - calls.fetch_add(1, Ordering::SeqCst); - takeover.notify_one(); - takeover_release.notified().await; - Ok(raw.as_ref().clone()) - }) - .await - } - }); - participants(content_id, 2).await; - leader.abort(); - assert!(leader.await.err().unwrap().is_cancelled()); - started(&takeover).await; - assert_eq!(calls.load(Ordering::SeqCst), 2); - assert_uncached(content_id); - takeover_release.notify_one(); - let projection = waiter.await.unwrap().unwrap(); - assert_eq!(projection.chunk_bytes(0).unwrap(), raw.as_slice()); - assert_flight_released(content_id); -} - -#[tokio::test] -async fn cancelled_waiter_does_not_cancel_or_leak_the_leader() { - let (content_id, raw) = unique_data(127); - let entered = Arc::new(Notify::new()); - let release = Arc::new(Notify::new()); - let leader = tokio::spawn({ - let entered = entered.clone(); - let release = release.clone(); - async move { - get_or_project(content_id, || async move { - entered.notify_one(); - release.notified().await; - Ok(raw.as_ref().clone()) - }) - .await - } - }); - started(&entered).await; - let waiter = tokio::spawn(async move { - get_or_project(content_id, || async { - panic!("cancelled waiter must not load while leader is active"); - }) - .await - }); - participants(content_id, 2).await; - waiter.abort(); - assert!(waiter.await.err().unwrap().is_cancelled()); - participants(content_id, 1).await; - release.notify_one(); - let projection = leader.await.unwrap().unwrap(); - assert_eq!(projection.map.file_content_id, content_id); - assert_flight_released(content_id); -} - -#[tokio::test] -async fn evicted_result_survives_leader_drop_until_joined_waiter_finishes() { - let (content_id, raw) = unique_data(131); - let calls = Arc::new(AtomicUsize::new(0)); - let entered = Arc::new(Notify::new()); - let release = Arc::new(Notify::new()); - let leader = tokio::spawn({ - let raw = raw.clone(); - let calls = calls.clone(); - let entered = entered.clone(); - let release = release.clone(); - async move { - get_or_project(content_id, || async move { - calls.fetch_add(1, Ordering::SeqCst); - entered.notify_one(); - release.notified().await; - Ok(raw.as_ref().clone()) - }) - .await - } - }); - started(&entered).await; - let registry = FLIGHTS.get().unwrap(); - let retained = FlightClaim::acquire(registry, content_id).unwrap(); - let result_weak; - { - // Queue a test-only lock before the real waiter, keeping its turn - // blocked until eviction and the leader's return owner are gone. - let queued = retained.flight.as_ref().unwrap().result.lock(); - tokio::pin!(queued); - assert!(futures::poll!(&mut queued).is_pending()); - let waiter = tokio::spawn({ - let raw = raw.clone(); - let calls = calls.clone(); - async move { - get_or_project(content_id, || async move { - calls.fetch_add(1, Ordering::SeqCst); - Ok(raw.as_ref().clone()) - }) - .await - } - }); - participants(content_id, 3).await; - release.notify_one(); - let projection = leader.await.unwrap().unwrap(); - let guard = queued.await; - result_weak = Arc::downgrade(&projection); - evict(content_id); - drop(projection); - assert_uncached(content_id); - assert!(result_weak.upgrade().is_some()); - drop(guard); - let shared = waiter.await.unwrap().unwrap(); - assert!(result_weak.ptr_eq(&Arc::downgrade(&shared))); - assert_eq!(shared.chunk_bytes(0).unwrap(), raw.as_slice()); - assert_eq!(calls.load(Ordering::SeqCst), 1); - drop(shared); - } - assert!(result_weak.upgrade().is_some()); - drop(retained); - assert!(result_weak.upgrade().is_none()); - assert_flight_released(content_id); - assert_uncached(content_id); - let reloaded = get_or_project(content_id, || async { - calls.fetch_add(1, Ordering::SeqCst); - Ok(raw.as_ref().clone()) - }) - .await - .unwrap(); - assert_eq!(calls.load(Ordering::SeqCst), 2); - assert_eq!(reloaded.chunk_bytes(0).unwrap(), raw.as_slice()); - assert_flight_released(content_id); -} - -#[test] -fn distinct_flights_are_bounded_but_existing_digest_can_join_at_capacity() { - let registry = Mutex::new(FlightRegistry::default()); - let mut claims = Vec::new(); - for index in 0..PROJECTION_FLIGHT_CAP { - claims.push(FlightClaim::acquire(®istry, [index as u8; 32]).unwrap()); - } - let joined = FlightClaim::acquire(®istry, [42; 32]).unwrap(); - assert!(Arc::ptr_eq( - claims[42].flight.as_ref().unwrap(), - joined.flight.as_ref().unwrap() - )); - let error = FlightClaim::acquire(®istry, [255; 32]).err().unwrap(); - assert_eq!(error.code, SnapshotErrorCode::LimitExceeded); - drop(claims); - assert_eq!(registry.lock().unwrap().entries.len(), 1); - drop(joined); - assert!(registry.lock().unwrap().entries.is_empty()); - let fresh = FlightClaim::acquire(®istry, [255; 32]).unwrap(); - drop(fresh); - assert!(registry.lock().unwrap().entries.is_empty()); -} - -#[test] -fn last_claim_removes_only_its_exact_registered_gate() { - let registry = Mutex::new(FlightRegistry::default()); - let content_id = [91; 32]; - let old = FlightClaim::acquire(®istry, content_id).unwrap(); - let replacement = Arc::new(ProjectionFlight { - result: tokio::sync::Mutex::new(None), - }); - registry - .lock() - .unwrap() - .entries - .insert(content_id, Arc::downgrade(&replacement)); - drop(old); - assert!(registry.lock().unwrap().entries[&content_id].ptr_eq(&Arc::downgrade(&replacement))); - let joined = FlightClaim::acquire(®istry, content_id).unwrap(); - assert!(Arc::ptr_eq(joined.flight.as_ref().unwrap(), &replacement)); - drop(replacement); - drop(joined); - assert!(registry.lock().unwrap().entries.is_empty()); - registry - .lock() - .unwrap() - .entries - .insert(content_id, Weak::new()); - let reclaimed = FlightClaim::acquire(®istry, content_id).unwrap(); - drop(reclaimed); - assert!(registry.lock().unwrap().entries.is_empty()); -} - -#[test] -fn concurrent_final_claim_drops_release_registry_capacity() { - let registry = Mutex::new(FlightRegistry::default()); - for index in 0..64 { - let content_id = [index; 32]; - let first = FlightClaim::acquire(®istry, content_id).unwrap(); - let second = FlightClaim::acquire(®istry, content_id).unwrap(); - let barrier = Barrier::new(2); - std::thread::scope(|scope| { - let barrier = &barrier; - scope.spawn(move || { - barrier.wait(); - drop(first); - }); - scope.spawn(move || { - barrier.wait(); - drop(second); - }); - }); - assert!(registry.lock().unwrap().entries.is_empty()); - } -} diff --git a/src/ceres/snapshot/chunks.rs b/src/ceres/snapshot/chunks.rs index 8999a45e..8346cec2 100644 --- a/src/ceres/snapshot/chunks.rs +++ b/src/ceres/snapshot/chunks.rs @@ -1,34 +1,132 @@ -//! On-the-fly, range-readable chunk projection for one file (spec 07). -//! -//! Persistent segment/locator storage is T04/T12 work; this identity slice -//! builds the MCM2 map and MCL2 leaves from a fixed view's verified blob and -//! retains inline bytes through 512 MiB and only verified map metadata for -//! larger files. Large CHUNK reads use the current request's strict raw range -//! source. This is a reproducible process cache, not a durable locator. -//! -//! The cache is an optimization, never the authority: every projected file -//! is re-hashed against the `content_id` the fixed view advertised, and a -//! cache miss simply rebuilds from Git. Entries are addressed by content -//! digest, never by request path. +//! Cold full-stream chunk-map verifier. Actual content callers use immutable +//! source receipts and selected persisted pages, never a digest-only cache. -use std::{ - collections::{HashMap, VecDeque}, - sync::{Arc, Mutex, OnceLock, Weak}, -}; +use std::sync::Arc; use mst2_codec::chunkmap::{CHUNK_SIZE, CHUNKS_PER_PAGE, ChunkLeaf, ChunkMap}; use sha2::{Digest, Sha256}; -use super::content_budget::{MemoryLease, projection_budget}; +use super::content_budget::MemoryLease; use crate::ceres::snapshot::error::{SnapshotError, SnapshotErrorCode}; #[path = "chunks_stream.rs"] mod streaming; -pub use streaming::get_or_project_stream; +/// Full-stream admission for one exact source, independent of the digest cache. +/// This opaque value is the only production input to durable map installation. +pub(crate) struct VerifiedSourceChunkMap { + source: ChunkMapSource, + projection: ChunkProjection, +} + +#[derive(Debug, Clone, PartialEq, Eq)] +pub(crate) struct ChunkMapSource { + fact: crate::callisto::mst2_verified_object::Model, +} + +impl ChunkMapSource { + pub(crate) fn from_fact( + fact: crate::callisto::mst2_verified_object::Model, + oid: &str, + ) -> Result { + if fact.id <= 0 + || fact.storage_domain != "git" + || fact.object_kind != "blob" + || fact.git_oid != oid + || ![40, 64].contains(&oid.len()) + || !oid + .bytes() + .all(|b| b.is_ascii_hexdigit() && !b.is_ascii_uppercase()) + || fact.state != "VERIFIED" + || fact.verification_version + != crate::jupiter::storage::mono_storage::MST2_VERIFICATION_VERSION + || fact.raw_sha256.len() != 32 + || fact.size <= 0 + || fact.size as u64 > 8 * 1024 * 1024 * 1024 * 1024 + { + return Err(SnapshotError::new( + SnapshotErrorCode::IntegrityError, + "invalid exact chunk map source fact", + )); + } + Ok(Self { fact }) + } + + pub(crate) fn fact(&self) -> &crate::callisto::mst2_verified_object::Model { + &self.fact + } + + pub(crate) fn canonical_bytes(&self) -> Result, SnapshotError> { + let f = &self.fact; + serde_json::to_vec(&( + f.id, + &f.storage_domain, + &f.git_oid, + &f.object_kind, + &f.raw_sha256, + f.size, + f.verification_version, + &f.state, + f.created_at, + )) + .map_err(|_| internal("chunk map source encoding failed")) + } +} + +impl VerifiedSourceChunkMap { + pub(crate) async fn verify( + handler: &T, + source: ChunkMapSource, + budget: &Arc, + ) -> Result { + let digest: [u8; 32] = source + .fact + .raw_sha256 + .as_slice() + .try_into() + .map_err(|_| internal("chunk map source digest shape changed"))?; + let projection = streaming::build_source_stream( + digest, + source.fact.size as u64, + || async { + handler + .get_raw_blob_stream_by_hash(&source.fact.git_oid) + .await + .map_err(|error| { + let code = match error { + crate::common::errors::MegaError::ObjStorageNotFound(_) => { + SnapshotErrorCode::ObjectUnavailable + } + crate::common::errors::MegaError::ObjStorageInconsistent(_) => { + SnapshotErrorCode::IntegrityError + } + _ => SnapshotErrorCode::Internal, + }; + SnapshotError::new(code, "exact chunk map source could not be read") + }) + }, + budget, + ) + .await?; + Ok(Self { source, projection }) + } + + pub(crate) fn source(&self) -> &ChunkMapSource { + &self.source + } + pub(crate) fn map(&self) -> &ChunkMap { + &self.projection.map + } + pub(crate) fn leaves(&self) -> &[ChunkLeaf] { + &self.projection.leaves + } + pub(crate) fn leaf_hashes(&self) -> &[[u8; 32]] { + &self.projection.leaf_hashes + } +} -pub(crate) fn projection_reservation_bytes(size: u64) -> Result { - streaming::reserved_bytes(size) +pub(crate) fn map_build_reservation_bytes(size: u64) -> Result { + streaming::reserved_map_bytes(size) } /// One file's verified range-readable projection. @@ -111,6 +209,7 @@ impl ChunkProjection { }) } + #[cfg(test)] pub fn has_inline_bytes(&self) -> bool { !self.raw.is_empty() } @@ -133,6 +232,7 @@ impl ChunkProjection { .sum::() } + #[cfg(test)] pub fn verify_chunk(&self, index: u64, bytes: &[u8]) -> Result<(), SnapshotError> { let want_len = self.map.chunk_len(index).map_err(codec_err)?; if bytes.len() as u64 != want_len { @@ -153,11 +253,13 @@ impl ChunkProjection { Ok(()) } + #[cfg(test)] pub fn page_count(&self) -> u64 { self.map.page_count } /// One MCL2 leaf plus its bottom-up proof toward `pages_root`. + #[cfg(test)] pub fn leaf_and_proof( &self, page_index: u64, @@ -179,6 +281,7 @@ impl ChunkProjection { /// Raw bytes of chunk `index`, length-checked against the map (spec 07 /// §4: full 1 MiB chunks except the positive-length remainder). + #[cfg(test)] pub fn chunk_bytes(&self, index: u64) -> Result<&[u8], SnapshotError> { let want_len = self.map.chunk_len(index).map_err(codec_err)?; let start = (index as usize) @@ -222,220 +325,6 @@ fn internal(m: &str) -> SnapshotError { SnapshotError::new(SnapshotErrorCode::Internal, m) } -/// Bounded process-wide cache of verified projections. Eviction is pure -/// memory reclaim; the projection is reproducible from Git, so an evicted -/// entry is rebuilt, never an error to the client. -static STAGED: OnceLock> = OnceLock::new(); - -/// 512 MiB cap for staged file bytes in this identity slice. The real -/// persistent RAW_OBJECT locator layout replaces this (spec 10 §4). -const STAGED_CAP_BYTES: usize = 512 * 1024 * 1024; -const STAGED_MAX_ENTRIES: usize = 4096; - -struct ProjectionCache { - entries: HashMap<[u8; 32], std::sync::Arc>, - /// Insertion/last-used order for FIFO reclaim. - order: VecDeque<[u8; 32]>, - total_bytes: usize, -} - -impl ProjectionCache { - fn new() -> Self { - ProjectionCache { - entries: HashMap::new(), - order: VecDeque::new(), - total_bytes: 0, - } - } - - fn get(&mut self, id: [u8; 32]) -> Option> { - self.entries.get(&id).cloned() - } - - fn evict_oldest(&mut self) -> Option> { - while let Some(victim) = self.order.pop_front() { - if let Some(projection) = self.entries.remove(&victim) { - self.total_bytes -= projection.retained_bytes(); - return Some(projection); - } - } - None - } - - fn put(&mut self, proj: std::sync::Arc) -> Vec> { - let mut evicted = Vec::new(); - let id = proj.map.file_content_id; - if self.entries.contains_key(&id) { - return evicted; - } - // Reclaim oldest entries until the new one fits. A file larger than - // the cap alone is not cached between flights but still - // served correctly. - let bytes = proj.retained_bytes(); - if bytes > STAGED_CAP_BYTES { - return evicted; - } - while (self.total_bytes + bytes > STAGED_CAP_BYTES - || self.entries.len() >= STAGED_MAX_ENTRIES) - && let Some(victim) = self.evict_oldest() - { - evicted.push(victim); - } - self.total_bytes += bytes; - self.order.push_back(id); - self.entries.insert(id, proj); - evicted - } -} - -static FLIGHTS: OnceLock> = OnceLock::new(); -const PROJECTION_FLIGHT_CAP: usize = 128; - -struct ProjectionFlight { - result: tokio::sync::Mutex>>, -} - -#[derive(Default)] -struct FlightRegistry { - entries: HashMap<[u8; 32], Weak>, -} - -struct FlightClaim<'a> { - content_id: [u8; 32], - flight: Option>, - registry: &'a Mutex, -} - -impl<'a> FlightClaim<'a> { - fn acquire( - registry: &'a Mutex, - content_id: [u8; 32], - ) -> Result { - let mut state = registry - .lock() - .map_err(|_| internal("chunk projection flight registry lock poisoned"))?; - if let Some(flight) = state.entries.get(&content_id).and_then(Weak::upgrade) { - return Ok(Self { - content_id, - flight: Some(flight), - registry, - }); - } - state.entries.remove(&content_id); - if state.entries.len() >= PROJECTION_FLIGHT_CAP { - return Err(SnapshotError::new( - SnapshotErrorCode::LimitExceeded, - "too many distinct chunk projections in flight", - )); - } - let flight = Arc::new(ProjectionFlight { - result: tokio::sync::Mutex::new(None), - }); - state.entries.insert(content_id, Arc::downgrade(&flight)); - Ok(Self { - content_id, - flight: Some(flight), - registry, - }) - } -} - -impl Drop for FlightClaim<'_> { - fn drop(&mut self) { - let Some(flight) = self.flight.take() else { - return; - }; - let Ok(mut state) = self.registry.lock() else { - return; - }; - if Arc::strong_count(&flight) == 1 { - if state - .entries - .get(&self.content_id) - .is_some_and(|registered| registered.ptr_eq(&Arc::downgrade(&flight))) - { - state.entries.remove(&self.content_id); - } - drop(state); - drop(flight); - } else { - // Serialize owner release with admission and other final drops. - // Another participant keeps the potentially large result alive. - drop(flight); - drop(state); - } - } -} - -/// Return the cached projection, or build one via `load` (which must resolve -/// the file in the fixed view and return its verified bytes). Concurrent -/// misses for the same digest share one successful load and projection. -#[cfg(test)] -pub async fn get_or_project( - content_id: [u8; 32], - load: F, -) -> Result, SnapshotError> -where - F: FnOnce() -> Fut, - Fut: std::future::Future, SnapshotError>>, -{ - get_or_project_with(content_id, || async { - ChunkProjection::build(content_id, load().await?) - }) - .await -} - -async fn get_or_project_with( - content_id: [u8; 32], - build: F, -) -> Result, SnapshotError> -where - F: FnOnce() -> Fut, - Fut: std::future::Future>, -{ - let cache = STAGED.get_or_init(|| Mutex::new(ProjectionCache::new())); - if let Some(p) = cache - .lock() - .map_err(|_| internal("chunk projection cache lock poisoned"))? - .get(content_id) - { - return Ok(p); - } - - let registry = FLIGHTS.get_or_init(|| Mutex::new(FlightRegistry::default())); - let claim = FlightClaim::acquire(registry, content_id)?; - let flight = claim - .flight - .as_ref() - .ok_or_else(|| internal("chunk projection flight claim released"))?; - let mut result = flight.result.lock().await; - if let Some(p) = cache - .lock() - .map_err(|_| internal("chunk projection cache lock poisoned"))? - .get(content_id) - { - *result = Some(p.clone()); - return Ok(p); - } - if let Some(p) = result.as_ref() { - return Ok(p.clone()); - } - let proj = Arc::new(build().await?); - let evicted = cache - .lock() - .map_err(|_| internal("chunk projection cache lock poisoned"))? - .put(proj.clone()); - // Final projection/credit drops can be expensive and must not hold the - // process cache mutex. Eviction does not refund other Arc owners. - drop(evicted); - *result = Some(proj.clone()); - Ok(proj) -} - -#[cfg(test)] -#[path = "chunk_singleflight_tests.rs"] -mod singleflight_tests; - #[cfg(test)] mod tests { use super::*; diff --git a/src/ceres/snapshot/chunks_stream.rs b/src/ceres/snapshot/chunks_stream.rs index 56f007fd..ab19e5b5 100644 --- a/src/ceres/snapshot/chunks_stream.rs +++ b/src/ceres/snapshot/chunks_stream.rs @@ -1,3 +1,5 @@ +use std::sync::{Arc, OnceLock}; + use futures::StreamExt; use tokio::sync::Semaphore; @@ -7,9 +9,20 @@ use crate::orbit_api::object_storage::ObjectByteStream; const STREAM_ITEM_MAX_BYTES: usize = 8 * 1024 * 1024; const MAX_FILE_BYTES: u64 = 8 * 1024 * 1024 * 1024 * 1024; const MAX_BUILDERS: usize = 4; +#[cfg(test)] +const STAGED_CAP_BYTES: usize = 512 * 1024 * 1024; const CONSTRUCTION_ALLOWANCE: usize = 64 * 1024; +#[cfg(test)] pub(super) fn reserved_bytes(size: u64) -> Result { + reserved_bytes_for(size, size <= STAGED_CAP_BYTES as u64) +} + +pub(super) fn reserved_map_bytes(size: u64) -> Result { + reserved_bytes_for(size, false) +} + +fn reserved_bytes_for(size: u64, inline: bool) -> Result { if size == 0 || size > MAX_FILE_BYTES { return Err(SnapshotError::new( if size == 0 { @@ -22,11 +35,7 @@ pub(super) fn reserved_bytes(size: u64) -> Result { } let chunks = size.div_ceil(CHUNK_SIZE as u64); let pages = chunks.div_ceil(CHUNKS_PER_PAGE as u64); - let inline = if size <= STAGED_CAP_BYTES as u64 { - size - } else { - 0 - }; + let inline = if inline { size } else { 0 }; let bytes = chunks .checked_mul(32) .and_then(|n| { @@ -41,65 +50,50 @@ pub(super) fn reserved_bytes(size: u64) -> Result { Ok(bytes) } -fn reserve_projection( +#[cfg(test)] +async fn build_stream( + content_id: [u8; 32], size: u64, - budget: &Arc, - cache: &Mutex, -) -> Result { - let bytes = reserved_bytes(size)?; - loop { - match budget.reserve(bytes) { - Ok(lease) => return Ok(lease), - Err(error) => { - if error.code != SnapshotErrorCode::TemporaryUnavailable { - return Err(error); - } - let victim = cache - .lock() - .map_err(|_| internal("chunk projection cache lock poisoned"))? - .evict_oldest(); - let Some(victim) = victim else { - return Err(error); - }; - drop(victim); - } - } - } + input: ObjectByteStream, + lease: MemoryLease, +) -> Result { + build_stream_with_inline( + content_id, + size, + input, + lease, + size <= STAGED_CAP_BYTES as u64, + ) + .await } -pub async fn get_or_project_stream( +pub(super) async fn build_source_stream( content_id: [u8; 32], size: u64, open: F, -) -> Result, SnapshotError> + budget: &Arc, +) -> Result where F: FnOnce() -> Fut, Fut: std::future::Future>, { - // Profile arithmetic also runs on hits; fixed facts remain per-request. - reserved_bytes(size)?; - get_or_project_with(content_id, || async { - static BUILDERS: OnceLock> = OnceLock::new(); - build_with_resources( - content_id, - size, - open, - projection_budget(), - BUILDERS.get_or_init(|| Arc::new(Semaphore::new(MAX_BUILDERS))), - STAGED.get_or_init(|| Mutex::new(ProjectionCache::new())), - ) - .await - }) + static BUILDERS: OnceLock> = OnceLock::new(); + build_source_with_resources( + content_id, + size, + open, + budget, + BUILDERS.get_or_init(|| Arc::new(Semaphore::new(MAX_BUILDERS))), + ) .await } -async fn build_with_resources( +async fn build_source_with_resources( content_id: [u8; 32], size: u64, open: F, budget: &Arc, builders: &Arc, - cache: &Mutex, ) -> Result where F: FnOnce() -> Fut, @@ -108,20 +102,39 @@ where let _builder = builders.clone().try_acquire_owned().map_err(|_| { SnapshotError::new( SnapshotErrorCode::TemporaryUnavailable, - "chunk projection builders are occupied", + "chunk map builders are occupied", ) })?; - let lease = reserve_projection(size, budget, cache)?; - // Source body I/O starts only after builder and retained-memory credit. + let lease = budget.reserve(source_reservation_bytes(size)?)?; let input = open().await?; - build_stream(content_id, size, input, lease).await + build_stream_with_inline(content_id, size, input, lease, false).await } -async fn build_stream( +pub(super) fn source_reservation_bytes(size: u64) -> Result { + let map_bytes = reserved_map_bytes(size)?; + let pages = size + .div_ceil(CHUNK_SIZE as u64) + .div_ceil(CHUNKS_PER_PAGE as u64); + let nodes = pages + .checked_mul(2) + .and_then(|n| n.checked_sub(1)) + .and_then(|n| { + n.checked_mul(std::mem::size_of::() as u64) + }) + .and_then(|n| usize::try_from(n).ok()) + .ok_or_else(|| internal("chunk map install reservation overflow"))?; + map_bytes + .checked_add(nodes) + .and_then(|n| n.checked_add(4 * 1024 * 1024)) + .ok_or_else(|| internal("chunk map install reservation overflow")) +} + +async fn build_stream_with_inline( content_id: [u8; 32], size: u64, mut input: ObjectByteStream, lease: MemoryLease, + inline: bool, ) -> Result { let chunk_count = size.div_ceil(CHUNK_SIZE as u64); let page_count = chunk_count.div_ceil(CHUNKS_PER_PAGE as u64); @@ -129,7 +142,6 @@ async fn build_stream( let mut leaves = Vec::new(); let mut leaf_hashes = Vec::new(); let mut current = Vec::new(); - let inline = size <= STAGED_CAP_BYTES as u64; if inline { raw.try_reserve_exact(size as usize) .map_err(allocation_error)?; diff --git a/src/ceres/snapshot/chunks_stream_tests.rs b/src/ceres/snapshot/chunks_stream_tests.rs index 2ccebae8..2cf7fb53 100644 --- a/src/ceres/snapshot/chunks_stream_tests.rs +++ b/src/ceres/snapshot/chunks_stream_tests.rs @@ -14,6 +14,112 @@ fn stream(parts: Vec>) -> ObjectByteStream { Box::pin(futures::stream::iter(parts)) } +#[tokio::test] +async fn exact_source_builder_admits_all_install_workspace_before_open_and_cancellation_releases_it() + { + let raw = Bytes::from_static(b"exact source body"); + let digest: [u8; 32] = Sha256::digest(&raw).into(); + let weight = source_reservation_bytes(raw.len() as u64).unwrap(); + assert!(weight > 4 * 1024 * 1024); + let budget = MemoryBudget::new(weight); + let builders = Arc::new(Semaphore::new(1)); + let opens = Arc::new(AtomicUsize::new(0)); + let held = budget.reserve(weight).unwrap(); + let error = build_source_with_resources( + digest, + raw.len() as u64, + || async { + opens.fetch_add(1, Ordering::SeqCst); + Ok(stream(vec![Ok(raw.clone())])) + }, + &budget, + &builders, + ) + .await + .err() + .unwrap(); + assert_eq!(error.code, SnapshotErrorCode::TemporaryUnavailable); + assert_eq!(opens.load(Ordering::SeqCst), 0); + assert_eq!(builders.available_permits(), 1); + drop(held); + let held_builder = builders.clone().acquire_owned().await.unwrap(); + assert_eq!( + build_source_with_resources( + digest, + raw.len() as u64, + || async { + opens.fetch_add(1, Ordering::SeqCst); + Ok(stream(vec![Ok(raw.clone())])) + }, + &budget, + &builders + ) + .await + .err() + .unwrap() + .code, + SnapshotErrorCode::TemporaryUnavailable + ); + assert_eq!(opens.load(Ordering::SeqCst), 0); + assert_eq!(budget.used(), 0); + drop(held_builder); + let entered = Arc::new(Notify::new()); + let task_budget = budget.clone(); + let task_builders = builders.clone(); + let task_entered = entered.clone(); + let task_opens = opens.clone(); + let drops = Arc::new(AtomicUsize::new(0)); + let producer_owner = DropCount(drops.clone()); + let size = raw.len() as u64; + let task = tokio::spawn(async move { + build_source_with_resources( + digest, + size, + || async { + task_opens.fetch_add(1, Ordering::SeqCst); + task_entered.notify_one(); + Ok(Box::pin(futures::stream::unfold( + producer_owner, + |owner| async move { + futures::future::pending::<()>().await; + Some((Ok(Bytes::new()), owner)) + }, + )) as ObjectByteStream) + }, + &task_budget, + &task_builders, + ) + .await + }); + timeout(Duration::from_secs(5), entered.notified()) + .await + .unwrap(); + assert_eq!(budget.used(), weight); + assert_eq!(builders.available_permits(), 0); + task.abort(); + assert!(task.await.err().unwrap().is_cancelled()); + assert_eq!(budget.used(), 0); + assert_eq!(builders.available_permits(), 1); + assert_eq!(drops.load(Ordering::SeqCst), 1); + let projection = build_source_with_resources( + digest, + size, + || async { Ok(stream(vec![Ok(raw.clone())])) }, + &budget, + &builders, + ) + .await + .unwrap(); + assert!(!projection.has_inline_bytes()); + assert_eq!(projection.map.file_content_id, digest); + assert_eq!(opens.load(Ordering::SeqCst), 2); + assert_eq!(builders.available_permits(), 1); + assert_eq!(budget.used(), weight); + projection.verify_chunk(0, &raw).unwrap(); + drop(projection); + assert_eq!(budget.used(), 0); +} + async fn project(raw: &[u8], parts: Vec>) -> ChunkProjection { let budget = MemoryBudget::new(reserved_bytes(raw.len() as u64).unwrap()); let lease = budget @@ -205,36 +311,6 @@ async fn visible_producer_item_limit_does_not_collect_a_large_item() { assert_eq!(budget.used(), 0); } -#[tokio::test] -async fn evicted_projection_remains_charged_until_last_reader_drops() { - let raw = b"owned credits"; - let budget = MemoryBudget::new(reserved_bytes(raw.len() as u64).unwrap()); - let lease = budget - .reserve(reserved_bytes(raw.len() as u64).unwrap()) - .unwrap(); - let projection = Arc::new( - build_stream( - Sha256::digest(raw).into(), - raw.len() as u64, - stream(vec![Ok(Bytes::from_static(raw))]), - lease, - ) - .await - .unwrap(), - ); - let weight = projection.retained_bytes(); - let mut cache = ProjectionCache::new(); - assert!(cache.put(projection.clone()).is_empty()); - let reader = projection.clone(); - drop(projection); - drop(cache.evict_oldest()); - assert_eq!(cache.total_bytes, 0); - assert_eq!(budget.used(), weight); - assert!(budget.reserve(1).is_err()); - drop(reader); - assert_eq!(budget.used(), 0); -} - struct DropCount(Arc); impl Drop for DropCount { fn drop(&mut self) { @@ -242,139 +318,6 @@ impl Drop for DropCount { } } -#[tokio::test] -async fn production_admission_rejects_live_credit_and_builder_overload_before_open() { - let weight = reserved_bytes(3).unwrap(); - let cache = Mutex::new(ProjectionCache::new()); - let budget = MemoryBudget::new(weight); - let builders = Arc::new(Semaphore::new(MAX_BUILDERS)); - let calls = AtomicUsize::new(0); - let held = budget.reserve(weight).unwrap(); - let error = build_with_resources( - Sha256::digest(b"abc").into(), - 3, - || async { - calls.fetch_add(1, Ordering::SeqCst); - Ok(stream(vec![Ok(Bytes::from_static(b"abc"))])) - }, - &budget, - &builders, - &cache, - ) - .await - .err() - .unwrap(); - assert_eq!(error.code, SnapshotErrorCode::TemporaryUnavailable); - assert_eq!(calls.load(Ordering::SeqCst), 0); - assert_eq!(builders.available_permits(), MAX_BUILDERS); - drop(held); - let mut workers = Vec::new(); - for _ in 0..MAX_BUILDERS { - workers.push(builders.clone().try_acquire_owned().unwrap()); - } - let error = build_with_resources( - Sha256::digest(b"abc").into(), - 3, - || async { - calls.fetch_add(1, Ordering::SeqCst); - Ok(stream(vec![Ok(Bytes::from_static(b"abc"))])) - }, - &budget, - &builders, - &cache, - ) - .await - .err() - .unwrap(); - assert_eq!(error.code, SnapshotErrorCode::TemporaryUnavailable); - assert_eq!(calls.load(Ordering::SeqCst), 0); - assert_eq!(budget.used(), 0); - drop(workers); - let projection = build_with_resources( - Sha256::digest(b"abc").into(), - 3, - || async { - calls.fetch_add(1, Ordering::SeqCst); - Ok(stream(vec![Ok(Bytes::from_static(b"abc"))])) - }, - &budget, - &builders, - &cache, - ) - .await - .unwrap(); - assert_eq!(calls.load(Ordering::SeqCst), 1); - assert_eq!(budget.used(), weight); - assert_eq!(builders.available_permits(), MAX_BUILDERS); - drop(projection); - assert_eq!(budget.used(), 0); -} - -#[tokio::test] -async fn cancelled_stream_owner_refunds_after_stream_drop_and_same_flight_waiter_takes_over() { - let raw = Bytes::from(uuid::Uuid::new_v4().as_bytes().to_vec()); - let id: [u8; 32] = Sha256::digest(&raw).into(); - let entered = Arc::new(Notify::new()); - let drops = Arc::new(AtomicUsize::new(0)); - let leader = tokio::spawn({ - let entered = entered.clone(); - let drops = drops.clone(); - async move { - get_or_project_stream(id, 16, || async { - let input: ObjectByteStream = Box::pin(futures::stream::unfold( - (entered, DropCount(drops)), - |(entered, owner)| async move { - entered.notify_one(); - std::future::pending::<()>().await; - Some((Ok(Bytes::new()), (entered, owner))) - }, - )); - Ok(input) - }) - .await - } - }); - timeout(Duration::from_secs(5), entered.notified()) - .await - .unwrap(); - let waiter = tokio::spawn(async move { - get_or_project_stream(id, 16, || async { - let input: ObjectByteStream = stream(vec![Ok(raw)]); - Ok(input) - }) - .await - }); - timeout(Duration::from_secs(5), async { - loop { - if FLIGHTS - .get() - .unwrap() - .lock() - .unwrap() - .entries - .get(&id) - .map_or(0, Weak::strong_count) - == 2 - { - break; - } - tokio::task::yield_now().await; - } - }) - .await - .unwrap(); - leader.abort(); - assert!(leader.await.err().unwrap().is_cancelled()); - assert_eq!(drops.load(Ordering::SeqCst), 1); - let projection = timeout(Duration::from_secs(5), waiter) - .await - .unwrap() - .unwrap() - .unwrap(); - assert_eq!(projection.map.file_content_id, id); - assert_eq!(projection.chunk_bytes(0).unwrap().len(), 16); -} - #[test] fn maximum_file_metadata_fits_and_protocol_or_quota_rejection_needs_no_source() { assert!(reserved_bytes(MAX_FILE_BYTES).unwrap() < 300 * 1024 * 1024); diff --git a/src/ceres/snapshot/mod.rs b/src/ceres/snapshot/mod.rs index 532d1da6..3fae7f32 100644 --- a/src/ceres/snapshot/mod.rs +++ b/src/ceres/snapshot/mod.rs @@ -3,6 +3,8 @@ //! Serving remains native-only. The independent namespace index is an unwired //! composition seam; identity, attestation and publication integration remain //! separate gates. Fixed-source readers must never look up current refs. +pub(crate) mod chunk_map_gate; +pub(crate) mod chunk_map_index; pub mod chunks; pub(crate) mod content_budget; pub mod descriptor; diff --git a/src/jupiter/migration/m20261008_000100_add_mst2_chunk_maps.rs b/src/jupiter/migration/m20261008_000100_add_mst2_chunk_maps.rs new file mode 100644 index 00000000..cd597a16 --- /dev/null +++ b/src/jupiter/migration/m20261008_000100_add_mst2_chunk_maps.rs @@ -0,0 +1,28 @@ +//! Append-only, independently authenticated chunk-map read indexes. + +use sea_orm::{ConnectionTrait, DbBackend}; +use sea_orm_migration::prelude::*; + +#[derive(DeriveMigrationName)] +pub struct Migration; + +#[async_trait::async_trait] +impl MigrationTrait for Migration { + async fn up(&self, manager: &SchemaManager) -> Result<(), DbErr> { + if manager.get_database_backend() != DbBackend::Postgres { + return Err(DbErr::Custom( + "persisted chunk maps require primary PostgreSQL".into(), + )); + } + manager + .get_connection() + .execute_unprepared(include_str!("m20261008_000100_chunk_maps.sql")) + .await?; + Ok(()) + } + + async fn down(&self, _manager: &SchemaManager) -> Result<(), DbErr> { + // No collector or migration rollback may silently discard receipts. + Ok(()) + } +} diff --git a/src/jupiter/migration/m20261008_000100_chunk_maps.sql b/src/jupiter/migration/m20261008_000100_chunk_maps.sql new file mode 100644 index 00000000..75b452d0 --- /dev/null +++ b/src/jupiter/migration/m20261008_000100_chunk_maps.sql @@ -0,0 +1,90 @@ +CREATE TABLE mst2_chunk_map ( + map_id bytea PRIMARY KEY CHECK (pg_catalog.octet_length(map_id)=32), + descriptor bytea NOT NULL CHECK (pg_catalog.octet_length(descriptor)=100), + page_count integer NOT NULL CHECK (page_count BETWEEN 1 AND 32768), + pages_root bytea NOT NULL CHECK (pg_catalog.octet_length(pages_root)=32) +); +CREATE TABLE mst2_chunk_map_leaf ( + map_id bytea NOT NULL REFERENCES mst2_chunk_map(map_id), + page_index integer NOT NULL CHECK (page_index BETWEEN 0 AND 32767), + payload bytea NOT NULL CHECK (pg_catalog.octet_length(payload) BETWEEN 48 AND 8208), + PRIMARY KEY(map_id,page_index) +); +CREATE TABLE mst2_chunk_map_node ( + map_id bytea NOT NULL REFERENCES mst2_chunk_map(map_id), + first_page integer NOT NULL CHECK (first_page BETWEEN 0 AND 32767), + page_count integer NOT NULL CHECK (page_count BETWEEN 1 AND 32768), + digest bytea NOT NULL CHECK (pg_catalog.octet_length(digest)=32), + PRIMARY KEY(map_id,first_page,page_count), + CHECK (first_page+page_count<=32768) +); +CREATE TABLE mst2_chunk_map_source ( + storage_domain text NOT NULL CHECK (storage_domain='git'), + git_oid text NOT NULL CHECK (git_oid ~ '^([0-9a-f]{40}|[0-9a-f]{64})$'), + object_kind text NOT NULL CHECK (object_kind='blob'), + fact_id bigint NOT NULL CHECK (fact_id>0), + source_id bytea NOT NULL UNIQUE CHECK (pg_catalog.octet_length(source_id)=32), + source_bytes bytea NOT NULL CHECK (pg_catalog.octet_length(source_bytes) BETWEEN 1 AND 2048), + primary_scope bytea NOT NULL CHECK (pg_catalog.octet_length(primary_scope) BETWEEN 1 AND 1024), + map_id bytea NOT NULL REFERENCES mst2_chunk_map(map_id), + receipt_digest bytea NOT NULL CHECK (pg_catalog.octet_length(receipt_digest)=32), + PRIMARY KEY(storage_domain,git_oid,object_kind) +); + +-- These rows are indexes, not proof that any source body was consumed. The +-- trusted object writer alone publishes the independent immutable receipt. +CREATE FUNCTION mst2_chunk_map_immutable() RETURNS trigger LANGUAGE plpgsql AS $$ +BEGIN RAISE EXCEPTION 'chunk map indexes and source receipts are append-only'; END $$; + +CREATE FUNCTION mst2_chunk_map_primary() RETURNS trigger LANGUAGE plpgsql AS $$ +BEGIN + IF pg_catalog.current_schema() IS DISTINCT FROM TG_TABLE_SCHEMA + OR pg_catalog.pg_is_in_recovery() + OR pg_catalog.current_setting('transaction_isolation')<>'read committed' THEN + RAISE EXCEPTION 'chunk map insertion requires its actual primary schema and READ COMMITTED'; + END IF; + RETURN NULL; +END $$; + +CREATE FUNCTION mst2_chunk_map_complete() RETURNS trigger LANGUAGE plpgsql AS $$ +DECLARE m record; leaves bigint; nodes bigint; root bytea; first_leaf integer; last_leaf integer; +BEGIN + EXECUTE pg_catalog.format('SELECT * FROM %I.mst2_chunk_map WHERE map_id=$1',TG_TABLE_SCHEMA) + INTO m USING NEW.map_id; + EXECUTE pg_catalog.format('SELECT pg_catalog.count(*),pg_catalog.min(page_index),pg_catalog.max(page_index) FROM %I.mst2_chunk_map_leaf WHERE map_id=$1',TG_TABLE_SCHEMA) + INTO leaves,first_leaf,last_leaf USING NEW.map_id; + EXECUTE pg_catalog.format('SELECT pg_catalog.count(*) FROM %I.mst2_chunk_map_node WHERE map_id=$1',TG_TABLE_SCHEMA) + INTO nodes USING NEW.map_id; + EXECUTE pg_catalog.format('SELECT digest FROM %I.mst2_chunk_map_node WHERE map_id=$1 AND first_page=0 AND page_count=$2',TG_TABLE_SCHEMA) + INTO root USING NEW.map_id,m.page_count; + IF m.map_id IS NULL OR leaves<>m.page_count OR first_leaf<>0 OR last_leaf<>m.page_count-1 + OR nodes<>2*m.page_count-1 OR root IS DISTINCT FROM m.pages_root + OR pg_catalog.substring(m.descriptor,69,32) IS DISTINCT FROM m.pages_root + OR pg_catalog.sha256(pg_catalog.convert_to('mega.mst2.chunkmap','UTF8')||pg_catalog.decode('00','hex')||m.descriptor) IS DISTINCT FROM m.map_id THEN + RAISE EXCEPTION 'chunk map installation is incomplete or inconsistent'; + END IF; + RETURN NULL; +END $$; + +CREATE FUNCTION mst2_chunk_map_index_insert() RETURNS trigger LANGUAGE plpgsql AS $$ +DECLARE admitted boolean; +BEGIN + EXECUTE pg_catalog.format('SELECT EXISTS(SELECT 1 FROM %I.mst2_chunk_map_source WHERE map_id=$1)',TG_TABLE_SCHEMA) + INTO admitted USING NEW.map_id; + IF admitted THEN RAISE EXCEPTION 'admitted chunk map index cannot acquire more rows'; END IF; + RETURN NEW; +END $$; + +DO $$ DECLARE t text; BEGIN + FOREACH t IN ARRAY ARRAY['mst2_chunk_map','mst2_chunk_map_leaf','mst2_chunk_map_node','mst2_chunk_map_source'] LOOP + EXECUTE pg_catalog.format('CREATE TRIGGER mst2_chunk_map_primary BEFORE INSERT ON %I FOR EACH STATEMENT EXECUTE FUNCTION mst2_chunk_map_primary()',t); + EXECUTE pg_catalog.format('CREATE TRIGGER mst2_chunk_map_immutable BEFORE UPDATE OR DELETE ON %I FOR EACH ROW EXECUTE FUNCTION mst2_chunk_map_immutable()',t); + EXECUTE pg_catalog.format('CREATE TRIGGER mst2_chunk_map_no_truncate BEFORE TRUNCATE ON %I FOR EACH STATEMENT EXECUTE FUNCTION mst2_chunk_map_immutable()',t); + END LOOP; +END $$; +CREATE CONSTRAINT TRIGGER mst2_chunk_map_complete AFTER INSERT ON mst2_chunk_map_source + DEFERRABLE INITIALLY DEFERRED FOR EACH ROW EXECUTE FUNCTION mst2_chunk_map_complete(); +CREATE TRIGGER mst2_chunk_map_index_insert BEFORE INSERT ON mst2_chunk_map_leaf + FOR EACH ROW EXECUTE FUNCTION mst2_chunk_map_index_insert(); +CREATE TRIGGER mst2_chunk_map_index_insert BEFORE INSERT ON mst2_chunk_map_node + FOR EACH ROW EXECUTE FUNCTION mst2_chunk_map_index_insert(); diff --git a/src/jupiter/migration/mod.rs b/src/jupiter/migration/mod.rs index 2d667a39..f5981f02 100644 --- a/src/jupiter/migration/mod.rs +++ b/src/jupiter/migration/mod.rs @@ -151,6 +151,7 @@ mod m20261007_000300_add_mst2_metadata_lifetime_history; mod m20261007_000400_add_mst2_qualified_metadata_gc; mod m20261007_000500_add_mst2_install_capability; mod m20261007_000600_add_mst2_storage_routes; +mod m20261008_000100_add_mst2_chunk_maps; mod runner; pub use m20260905_000100_add_push_queue::ensure_queue_control_seed; pub use runner::apply_migrations; @@ -290,6 +291,7 @@ impl MigratorTrait for Migrator { Box::new(m20261007_000400_add_mst2_qualified_metadata_gc::Migration), Box::new(m20261007_000500_add_mst2_install_capability::Migration), Box::new(m20261007_000600_add_mst2_storage_routes::Migration), + Box::new(m20261008_000100_add_mst2_chunk_maps::Migration), ] } } diff --git a/src/jupiter/storage/mod.rs b/src/jupiter/storage/mod.rs index b130aa5f..2ea787de 100644 --- a/src/jupiter/storage/mod.rs +++ b/src/jupiter/storage/mod.rs @@ -18,6 +18,7 @@ pub mod media_paging_storage; pub mod mono_storage; pub(crate) mod mst2_publication_storage; pub mod mst2_retention; +pub(crate) mod native_chunk_map; pub mod native_metadata_install; pub(crate) mod native_publication_storage; pub(crate) mod native_snapshot_session; @@ -159,6 +160,8 @@ pub struct Storage { /// Derived native projection memoization, scoped to this storage assembly. /// Clones share it; independent databases/backends never share entries. pub(crate) native_projection_cache: Arc, + pub(crate) native_chunk_maps: + Arc>, pub(crate) native_snapshot_sessions: Arc>, pub(crate) projection_observation_sink: @@ -326,6 +329,7 @@ impl Storage { app_service: app_service.into(), native_projection_cache: Arc::default(), native_snapshot_sessions: Arc::default(), + native_chunk_maps: Arc::default(), projection_observation_sink: None, config_handle, config, @@ -704,6 +708,7 @@ impl Storage { app_service, native_projection_cache: Arc::default(), native_snapshot_sessions: Arc::default(), + native_chunk_maps: Arc::default(), projection_observation_sink: None, // app_service: AppService::mock(), cl_service: CLService::mock(), diff --git a/src/jupiter/storage/mono_storage.rs b/src/jupiter/storage/mono_storage.rs index 335869ac..5228c6fc 100644 --- a/src/jupiter/storage/mono_storage.rs +++ b/src/jupiter/storage/mono_storage.rs @@ -1789,12 +1789,43 @@ impl MonoStorage { if oids.is_empty() { return Ok(HashMap::new()); } - let rows = mst2_verified_object::Entity::find() - .filter(mst2_verified_object::Column::StorageDomain.eq("git")) - .filter(mst2_verified_object::Column::ObjectKind.eq("blob")) - .filter(mst2_verified_object::Column::GitOid.is_in(oids)) - .all(self.get_connection()) - .await?; + let connection = self.get_connection(); + let rows = if connection.get_database_backend() == sea_orm::DbBackend::Postgres { + use sea_orm::{DbBackend, FromQueryResult, Statement}; + let scope = connection.query_one_raw(Statement::from_string(DbBackend::Postgres, + "SELECT n.nspname AS schema FROM pg_catalog.pg_namespace n JOIN pg_catalog.pg_class c ON c.relnamespace=n.oid AND c.relname='mst2_verified_object' AND c.relkind='r' WHERE n.nspname=pg_catalog.current_schema() AND n.nspname NOT LIKE 'pg_temp_%'")) + .await?.ok_or_else(|| MegaError::Other("actual verified blob relation is missing".into()))?; + let schema: String = scope.try_get("", "schema")?; + let relation = format!("\"{}\".mst2_verified_object", schema.replace('"', "\"\"")); + let oids = serde_json::to_string(&oids) + .map_err(|error| MegaError::Other(error.to_string()))?; + let sql = format!( + "SELECT v.id,v.storage_domain,CASE WHEN pg_catalog.octet_length(v.git_oid) IN (40,64) THEN v.git_oid ELSE NULL END AS git_oid,v.object_kind,CASE WHEN pg_catalog.octet_length(v.raw_sha256)=32 THEN v.raw_sha256 ELSE NULL END AS raw_sha256,CASE WHEN v.size BETWEEN 0 AND {MST2_MAX_FILE_SIZE} THEN v.size ELSE NULL END AS size,CASE WHEN v.verification_version IN (1,{MST2_VERIFICATION_VERSION}) THEN v.verification_version ELSE NULL END AS verification_version,CASE WHEN v.state='VERIFIED' THEN v.state ELSE NULL END AS state,v.created_at FROM {relation} v WHERE v.storage_domain='git' AND v.object_kind='blob' AND v.git_oid IN (SELECT value FROM pg_catalog.jsonb_array_elements_text($1::jsonb))" + ); + connection + .query_all_raw(Statement::from_sql_and_values( + DbBackend::Postgres, + sql, + [oids.into()], + )) + .await? + .iter() + .map(|row| { + mst2_verified_object::Model::from_query_result(row, "").map_err(|_| { + MegaError::ObjStorageInconsistent( + "invalid bounded MST/2 verified blob record".into(), + ) + }) + }) + .collect::, _>>()? + } else { + mst2_verified_object::Entity::find() + .filter(mst2_verified_object::Column::StorageDomain.eq("git")) + .filter(mst2_verified_object::Column::ObjectKind.eq("blob")) + .filter(mst2_verified_object::Column::GitOid.is_in(oids)) + .all(connection) + .await? + }; for row in &rows { if row.state != "VERIFIED" || ![1, MST2_VERIFICATION_VERSION].contains(&row.verification_version) diff --git a/src/jupiter/storage/native_chunk_map.rs b/src/jupiter/storage/native_chunk_map.rs new file mode 100644 index 00000000..7dcbaa22 --- /dev/null +++ b/src/jupiter/storage/native_chunk_map.rs @@ -0,0 +1,684 @@ +//! Source-bound chunk-map receipts with indexed, authenticated selected pages. +//! +//! The object writer is trusted like the verified-object writer. A database +//! row, its checksum or a map's presence is never a full-source proof. Only +//! the opaque full-stream verifier can install a source receipt. This initial +//! store is append-only; bounded retention and collection remain separate work. + +use std::sync::Arc; + +use bytes::Bytes; +use futures::StreamExt; +use mst2_codec::chunkmap::{CHUNKS_PER_PAGE, ChunkLeaf, ChunkMap, ProofStep, verify_leaf}; +use sea_orm::{ + ConnectionTrait, DatabaseConnection, DatabaseTransaction, DbBackend, FromQueryResult, + IsolationLevel, QueryResult, Statement, TransactionTrait, +}; +use serde_json::json; +use sha2::{Digest, Sha256}; + +use super::object_storage::MegaObjectStorageWrapper; +use crate::{ + callisto::mst2_verified_object, + ceres::snapshot::{ + chunk_map_index::{ChunkMapNode, indexed_nodes, proof_intervals, selected_proof}, + chunks::{ChunkMapSource, VerifiedSourceChunkMap}, + content_budget::{MemoryBudget, MemoryLease, projection_budget}, + error::{SnapshotError, SnapshotErrorCode}, + }, + orbit_api::object_storage::{ObjectKey, ObjectMeta, ObjectNamespace}, +}; + +const DESCRIPTOR_CREDIT: usize = 32 * 1024; +const PAGE_CREDIT: usize = 64 * 1024; +const RECEIPT_DOMAIN: &[u8] = b"MST2-CHUNK-MAP-RECEIPT\0"; + +pub(crate) struct PostgresChunkMapRepository { + connection: DatabaseConnection, + primary_scope: Vec, + budget: Arc, + schema: String, +} + +pub(crate) struct PersistedChunkMap { + pub map: ChunkMap, + pub map_id: [u8; 32], + source: ChunkMapSource, + source_id: [u8; 32], + _memory: MemoryLease, +} + +impl PersistedChunkMap { + pub(crate) fn source_id(&self) -> [u8; 32] { + self.source_id + } +} + +pub(crate) struct AuthenticatedChunkPage { + pub leaf: ChunkLeaf, + pub proof: Vec, + _memory: MemoryLease, +} + +impl AuthenticatedChunkPage { + pub(crate) fn verify_chunk( + &self, + map: &ChunkMap, + index: u64, + bytes: &[u8], + ) -> Result<(), SnapshotError> { + let length = map + .chunk_len(index) + .map_err(|_| integrity("invalid requested chunk index"))?; + if index / CHUNKS_PER_PAGE as u64 != self.leaf.page_index || bytes.len() as u64 != length { + return Err(integrity( + "exact range length or selected chunk page disagrees", + )); + } + let slot = (index % CHUNKS_PER_PAGE as u64) as usize; + let expected = self + .leaf + .chunk_sha256 + .get(slot) + .ok_or_else(|| integrity("selected chunk digest is missing"))?; + let digest: [u8; 32] = Sha256::digest(bytes).into(); + if &digest != expected { + return Err(integrity( + "exact range digest disagrees with the authenticated selected page", + )); + } + Ok(()) + } +} + +impl PostgresChunkMapRepository { + pub(crate) async fn new(connection: DatabaseConnection) -> Result { + let schema = capture_schema(&connection).await?; + let primary_scope = read_scope(&connection, &schema).await?; + Ok(Self { + connection, + primary_scope, + budget: projection_budget().clone(), + schema, + }) + } + + pub(crate) async fn read( + &self, + source: &ChunkMapSource, + objects: &MegaObjectStorageWrapper, + ) -> Result>, SnapshotError> { + let memory = self.budget.reserve(DESCRIPTOR_CREDIT)?; + let txn = self.transaction().await?; + let result = async { + self.require_scope(&txn).await?; + require_current_source(&txn, source, &self.schema).await?; + let Some(row) = source_row(&txn, source, &self.schema).await? else { + return Ok(None); + }; + let (map, source_id) = self.validate_source_row(source, &row)?; + let bytes = receipt(&self.primary_scope, source, &map)?; + read_receipt(objects, &receipt_key(source_id), &bytes).await?; + Ok(Some(Arc::new(PersistedChunkMap { + map_id: map.map_id(), + map, + source: source.clone(), + source_id, + _memory: memory, + }))) + } + .await; + finish(txn, result).await + } + + /// Publication accepts no caller-provided map or "verified" flag. + pub(crate) async fn install( + &self, + verified: VerifiedSourceChunkMap, + objects: &MegaObjectStorageWrapper, + ) -> Result<(), SnapshotError> { + let source = verified.source(); + let map = verified.map(); + // The verifier reserved index and bounded SQL parameter workspace + // before opening the body. It owns those credits through publication. + let nodes = indexed_nodes(verified.leaf_hashes())?; + if nodes.last().map(|n| n.digest) != Some(map.pages_root) { + return Err(integrity( + "verified chunk map index disagrees with canonical root", + )); + } + let source_bytes = source.canonical_bytes()?; + let source_id = source_id(&self.primary_scope, &source_bytes); + let bytes = receipt(&self.primary_scope, source, map)?; + let key = receipt_key(source_id); + let txn = self.transaction().await?; + let result = async { + self.require_scope(&txn).await?; + require_current_source(&txn, source, &self.schema).await?; + if let Some(row) = source_row(&txn, source, &self.schema).await? { + let (stored, _) = self.validate_source_row(source, &row)?; + if stored != *map { return Err(integrity("immutable source map conflicts with reverified content")); } + read_receipt(objects, &key, &bytes).await?; + compare_complete_map(&txn, &verified, &nodes, &self.schema).await?; + return Ok(()); + } + // The complete source pass happened before this atomic create. + // A lost DB commit leaves an orphan trusted receipt, never an + // admitted partial map; exact re-verification can replay it. + objects.inner.put_metadata_atomic_create(&key, Bytes::copy_from_slice(&bytes), ObjectMeta { + size: bytes.len() as i64, ..Default::default() + }).await.map_err(storage_error)?; + read_receipt(objects, &key, &bytes).await?; + txn.execute_raw(stmt(&self.schema, "INSERT INTO mst2_chunk_map(map_id,descriptor,page_count,pages_root) VALUES($1,$2,$3,$4) ON CONFLICT(map_id) DO NOTHING", + [map.map_id().to_vec().into(), map.encode().into(), (map.page_count as i32).into(), map.pages_root.to_vec().into()])).await.map_err(db_error)?; + // Existing indexes must be compared, never repaired or overwritten. + let present = txn.query_one_raw(stmt(&self.schema, "SELECT EXISTS(SELECT 1 FROM mst2_chunk_map_source WHERE map_id=$1) AS sealed", [map.map_id().to_vec().into()])) + .await.map_err(db_error)?.ok_or_else(|| integrity("chunk map seal observation is missing"))? + .try_get::("", "sealed").map_err(db_error)?; + if !present { + for batch in verified.leaves().chunks(64) { + let encoded = encode_leaves(batch)?; + txn.execute_raw(stmt(&self.schema, "INSERT INTO mst2_chunk_map_leaf(map_id,page_index,payload) SELECT $1,p.page_index,pg_catalog.decode(p.payload,'hex') FROM pg_catalog.jsonb_to_recordset($2::jsonb) AS p(page_index integer,payload text) ON CONFLICT(map_id,page_index) DO NOTHING", + [map.map_id().to_vec().into(), encoded.into()])).await.map_err(db_error)?; + } + for batch in nodes.chunks(1024) { + let encoded = encode_nodes(batch)?; + txn.execute_raw(stmt(&self.schema, "INSERT INTO mst2_chunk_map_node(map_id,first_page,page_count,digest) SELECT $1,p.first_page,p.page_count,pg_catalog.decode(p.digest,'hex') FROM pg_catalog.jsonb_to_recordset($2::jsonb) AS p(first_page integer,page_count integer,digest text) ON CONFLICT(map_id,first_page,page_count) DO NOTHING", + [map.map_id().to_vec().into(), encoded.into()])).await.map_err(db_error)?; + } + } + compare_complete_map(&txn, &verified, &nodes, &self.schema).await?; + let fact = source.fact(); + txn.execute_raw(stmt(&self.schema, "INSERT INTO mst2_chunk_map_source(storage_domain,git_oid,object_kind,fact_id,source_id,source_bytes,primary_scope,map_id,receipt_digest) VALUES($1,$2,$3,$4,$5,$6,$7,$8,$9) ON CONFLICT(storage_domain,git_oid,object_kind) DO NOTHING", + [fact.storage_domain.clone().into(), fact.git_oid.clone().into(), fact.object_kind.clone().into(), fact.id.into(), + source_id.to_vec().into(), source_bytes.into(), self.primary_scope.clone().into(), map.map_id().to_vec().into(), Sha256::digest(&bytes).to_vec().into()])).await.map_err(db_error)?; + let row = source_row(&txn, source, &self.schema).await?.ok_or_else(|| integrity("installed source receipt is missing"))?; + let (stored, _) = self.validate_source_row(source, &row)?; + if stored != *map { return Err(integrity("concurrent source installation conflicts")); } + Ok(()) + }.await; + finish(txn, result).await + } + + pub(crate) async fn selected_page( + &self, + map: &PersistedChunkMap, + page_index: u64, + ) -> Result, SnapshotError> { + let intervals = proof_intervals(map.map.page_count, page_index)?; + let memory = self.budget.reserve(PAGE_CREDIT)?; + let txn = self.transaction().await?; + let result = async { + self.require_scope(&txn).await?; + require_current_source(&txn, &map.source, &self.schema).await?; + let row = txn.query_one_raw(stmt(&self.schema, "SELECT pg_catalog.octet_length(payload) AS size,CASE WHEN pg_catalog.octet_length(payload) BETWEEN 48 AND 8208 THEN payload ELSE NULL END AS payload FROM mst2_chunk_map_leaf WHERE map_id=$1 AND page_index=$2", + [map.map_id.to_vec().into(), (page_index as i32).into()])).await.map_err(db_error)? + .ok_or_else(|| integrity("admitted chunk map selected leaf is missing"))?; + let bytes = row.try_get::>>("", "payload").map_err(db_error)? + .ok_or_else(|| integrity("admitted chunk map selected leaf exceeds its byte profile"))?; + let leaf = ChunkLeaf::decode(&bytes).map_err(|_| integrity("admitted chunk map selected leaf is not canonical MCL2"))?; + if leaf.page_index != page_index || leaf.chunk_sha256.len() as u64 != ChunkLeaf::expected_count(map.map.chunk_count, page_index) + || leaf.encode().map_err(|_| integrity("invalid selected leaf"))? != bytes { + return Err(integrity("selected chunk map leaf identity or chunk count disagrees")); + } + let requested: Vec<_> = intervals.iter().map(|&(_, start, pages)| json!({"first_page":start,"page_count":pages})).collect(); + let encoded = serde_json::to_string(&requested).map_err(db_error)?; + let rows = txn.query_all_raw(stmt(&self.schema, "SELECT n.first_page,n.page_count,CASE WHEN pg_catalog.octet_length(n.digest)=32 THEN n.digest ELSE NULL END AS digest FROM pg_catalog.jsonb_to_recordset($2::jsonb) AS p(first_page integer,page_count integer) JOIN mst2_chunk_map_node n ON n.map_id=$1 AND n.first_page=p.first_page AND n.page_count=p.page_count", + [map.map_id.to_vec().into(), encoded.into()])).await.map_err(db_error)?; + let mut nodes = Vec::with_capacity(rows.len()); + for row in rows { + let digest: Option> = row.try_get("", "digest").map_err(db_error)?; + nodes.push(ChunkMapNode { + start: row.try_get::("", "first_page").map_err(db_error)? as u64, + pages: row.try_get::("", "page_count").map_err(db_error)? as u64, + digest: digest.ok_or_else(|| integrity("admitted chunk map sibling has invalid digest size"))?.as_slice().try_into().map_err(|_| integrity("invalid persisted sibling digest"))?, + }); + } + let proof = selected_proof(&intervals, &nodes)?; + verify_leaf(map.map.page_count, page_index, leaf.leaf_hash().map_err(|_| integrity("invalid leaf digest"))?, &proof, map.map.pages_root) + .map_err(|_| integrity("selected chunk map leaf or proof authentication failed"))?; + tracing::debug!(target:"mst2::chunk_map", selected_leaf_bytes = bytes.len(), selected_sibling_rows = nodes.len(), page_count = map.map.page_count, "authenticated persisted chunk map selected page"); + Ok(Arc::new(AuthenticatedChunkPage { leaf, proof, _memory: memory })) + }.await; + finish(txn, result).await + } + + async fn transaction(&self) -> Result { + let txn = self + .connection + .begin_with_config(Some(IsolationLevel::ReadCommitted), None) + .await + .map_err(db_error)?; + let search_path = format!("{},pg_catalog,pg_temp", quoted(&self.schema)); + txn.query_one_raw(Statement::from_sql_and_values( + DbBackend::Postgres, + "SELECT pg_catalog.set_config('search_path',$1,true)", + [search_path.into()], + )) + .await + .map_err(db_error)?; + txn.execute_unprepared("SET LOCAL statement_timeout='5s'; SET LOCAL lock_timeout='5s'") + .await + .map_err(db_error)?; + Ok(txn) + } + + async fn require_scope(&self, connection: &C) -> Result<(), SnapshotError> { + if read_scope(connection, &self.schema).await? != self.primary_scope { + return Err(integrity( + "chunk map repository no longer targets its captured primary scope", + )); + } + Ok(()) + } + + pub(crate) fn memory_budget(&self) -> &Arc { + &self.budget + } + + pub(crate) fn source_identity( + &self, + source: &ChunkMapSource, + ) -> Result<[u8; 32], SnapshotError> { + Ok(source_id(&self.primary_scope, &source.canonical_bytes()?)) + } + + #[cfg(test)] + pub(crate) fn with_test_budget(mut self, budget: Arc) -> Self { + self.budget = budget; + self + } + + #[cfg(test)] + pub(crate) fn test_primary_scope(&self) -> &[u8] { + &self.primary_scope + } + + fn validate_source_row( + &self, + source: &ChunkMapSource, + row: &QueryResult, + ) -> Result<(ChunkMap, [u8; 32]), SnapshotError> { + let source_bytes = source.canonical_bytes()?; + let source_id = source_id(&self.primary_scope, &source_bytes); + let source_bytes_row = bounded_bytes(row, "source_bytes")?; + let primary_scope_row = bounded_bytes(row, "primary_scope")?; + if row.try_get::("", "fact_id").map_err(db_error)? != source.fact().id + || source_bytes_row != source_bytes + || primary_scope_row != self.primary_scope + || bounded_bytes(row, "source_id")? != source_id + { + return Err(integrity( + "persisted map admission disagrees with the exact current source fact or primary", + )); + } + let bytes = row + .try_get::>>("", "descriptor") + .map_err(db_error)? + .ok_or_else(|| { + integrity("admitted chunk map descriptor is missing or outside its byte profile") + })?; + let map = ChunkMap::decode(&bytes) + .map_err(|_| integrity("admitted chunk map descriptor is invalid MCM2"))?; + let receipt_digest: [u8; 32] = + Sha256::digest(receipt(&self.primary_scope, source, &map)?).into(); + if map.encode() != bytes + || map.map_id().as_slice() != bounded_bytes(row, "map_id")?.as_slice() + || map.file_content_id.as_slice() != source.fact().raw_sha256.as_slice() + || map.file_size != source.fact().size as u64 + || map.page_count != row.try_get::("", "page_count").map_err(db_error)? as u64 + || map.pages_root.as_slice() != bounded_bytes(row, "pages_root")?.as_slice() + || bounded_bytes(row, "receipt_digest")? != receipt_digest + { + return Err(integrity( + "admitted chunk map descriptor or receipt identity disagrees", + )); + } + Ok((map, source_id)) + } +} + +async fn capture_schema(connection: &C) -> Result { + let row = connection.query_one_raw(Statement::from_string(DbBackend::Postgres, + "SELECT n.nspname AS schema FROM pg_catalog.pg_namespace n JOIN pg_catalog.pg_class c ON c.relnamespace=n.oid AND c.relname='mst2_metadata_storage_scope' AND c.relkind='r' WHERE n.nspname=pg_catalog.current_schema() AND n.nspname NOT LIKE 'pg_temp_%'")) + .await.map_err(db_error)?.ok_or_else(|| integrity("chunk map actual primary storage relation is missing"))?; + row.try_get("", "schema").map_err(db_error) +} + +async fn read_scope( + connection: &C, + schema: &str, +) -> Result, SnapshotError> { + if connection.get_database_backend() != DbBackend::Postgres { + return Err(integrity("chunk map repository requires PostgreSQL")); + } + let row = connection.query_one_raw(stmt(schema, "SELECT pg_catalog.pg_is_in_recovery() AS replica,CASE WHEN pg_catalog.octet_length(s.storage_uuid)<=255 THEN s.storage_uuid ELSE NULL END AS storage_uuid,pg_catalog.current_database() AS database,d.oid::bigint AS database_oid,n.nspname AS schema,n.oid::bigint AS schema_oid,pg_catalog.inet_server_addr()::text AS address,pg_catalog.inet_server_port() AS port FROM mst2_metadata_storage_scope s JOIN pg_catalog.pg_database d ON d.datname=pg_catalog.current_database() JOIN pg_catalog.pg_namespace n ON n.nspname=$1 JOIN pg_catalog.pg_class c ON c.relnamespace=n.oid AND c.relname='mst2_metadata_storage_scope' AND c.relkind='r' WHERE s.singleton=1", [schema.into()])) + .await.map_err(db_error)?.ok_or_else(|| integrity("chunk map primary scope is missing"))?; + if row.try_get::("", "replica").map_err(db_error)? { + return Err(integrity("chunk map repository is on a replica")); + } + let scope = serde_json::to_vec(&( + row.try_get::>("", "storage_uuid") + .map_err(db_error)? + .ok_or_else(|| integrity("chunk map primary scope UUID exceeds its bounded profile"))?, + row.try_get::("", "database").map_err(db_error)?, + row.try_get::("", "database_oid").map_err(db_error)?, + row.try_get::("", "schema").map_err(db_error)?, + row.try_get::("", "schema_oid").map_err(db_error)?, + row.try_get::>("", "address") + .map_err(db_error)?, + row.try_get::>("", "port").map_err(db_error)?, + )) + .map_err(db_error)?; + if scope.len() > 1024 { + return Err(integrity( + "chunk map primary scope exceeds its admitted profile", + )); + } + Ok(scope) +} + +async fn require_current_source( + connection: &C, + source: &ChunkMapSource, + schema: &str, +) -> Result<(), SnapshotError> { + let row = connection + .query_one_raw(stmt(schema, + "SELECT id,CASE WHEN storage_domain='git' THEN storage_domain ELSE NULL END AS storage_domain,CASE WHEN git_oid=$2 AND pg_catalog.octet_length(git_oid) IN (40,64) THEN git_oid ELSE NULL END AS git_oid,CASE WHEN object_kind='blob' THEN object_kind ELSE NULL END AS object_kind,CASE WHEN pg_catalog.octet_length(raw_sha256)=32 THEN raw_sha256 ELSE NULL END AS raw_sha256,CASE WHEN size BETWEEN 1 AND 8796093022208 THEN size ELSE NULL END AS size,CASE WHEN verification_version=2 THEN verification_version ELSE NULL END AS verification_version,CASE WHEN state='VERIFIED' THEN state ELSE NULL END AS state,created_at FROM mst2_verified_object WHERE id=$1 FOR SHARE", + [source.fact().id.into(), source.fact().git_oid.clone().into()], + )) + .await + .map_err(db_error)? + .ok_or_else(|| { + SnapshotError::new( + SnapshotErrorCode::MetadataNotReady, + "chunk map source has no current verified fact", + ) + })?; + let current = mst2_verified_object::Model::from_query_result(&row, "").map_err(|_| { + integrity("chunk map current source fact is outside its bounded verified profile") + })?; + if ¤t != source.fact() { + return Err(integrity( + "chunk map source fact changed during request or verification", + )); + } + Ok(()) +} + +async fn source_row( + connection: &C, + source: &ChunkMapSource, + schema: &str, +) -> Result, SnapshotError> { + connection.query_one_raw(stmt(schema, "SELECT s.fact_id,CASE WHEN pg_catalog.octet_length(s.source_id)=32 THEN s.source_id ELSE NULL END AS source_id,CASE WHEN pg_catalog.octet_length(s.source_bytes)<=2048 THEN s.source_bytes ELSE NULL END AS source_bytes,CASE WHEN pg_catalog.octet_length(s.primary_scope)<=1024 THEN s.primary_scope ELSE NULL END AS primary_scope,CASE WHEN pg_catalog.octet_length(s.map_id)=32 THEN s.map_id ELSE NULL END AS map_id,CASE WHEN pg_catalog.octet_length(s.receipt_digest)=32 THEN s.receipt_digest ELSE NULL END AS receipt_digest,CASE WHEN pg_catalog.octet_length(m.descriptor)=100 THEN m.descriptor ELSE NULL END AS descriptor,m.page_count,CASE WHEN pg_catalog.octet_length(m.pages_root)=32 THEN m.pages_root ELSE NULL END AS pages_root FROM mst2_chunk_map_source s LEFT JOIN mst2_chunk_map m ON m.map_id=s.map_id WHERE s.storage_domain=$1 AND s.git_oid=$2 AND s.object_kind=$3", + [source.fact().storage_domain.clone().into(), source.fact().git_oid.clone().into(), source.fact().object_kind.clone().into()])).await.map_err(db_error) +} + +fn source_id(scope: &[u8], source: &[u8]) -> [u8; 32] { + let mut hash = Sha256::new(); + hash.update(RECEIPT_DOMAIN); + hash.update((scope.len() as u32).to_be_bytes()); + hash.update(scope); + hash.update((source.len() as u32).to_be_bytes()); + hash.update(source); + hash.finalize().into() +} + +fn receipt( + scope: &[u8], + source: &ChunkMapSource, + map: &ChunkMap, +) -> Result, SnapshotError> { + let source = source.canonical_bytes()?; + if scope.len() > 1024 || source.len() > 2048 { + return Err(integrity( + "chunk map source receipt exceeds its fixed profile", + )); + } + let mut bytes = Vec::with_capacity(RECEIPT_DOMAIN.len() + 8 + scope.len() + source.len() + 100); + bytes.extend_from_slice(RECEIPT_DOMAIN); + bytes.extend_from_slice(&(scope.len() as u32).to_be_bytes()); + bytes.extend_from_slice(scope); + bytes.extend_from_slice(&(source.len() as u32).to_be_bytes()); + bytes.extend_from_slice(&source); + bytes.extend_from_slice(&map.encode()); + Ok(bytes) +} + +fn receipt_key(source_id: [u8; 32]) -> ObjectKey { + ObjectKey { + namespace: ObjectNamespace::ChunkMapReceipt, + key: hex::encode(source_id), + } +} + +async fn read_receipt( + objects: &MegaObjectStorageWrapper, + key: &ObjectKey, + expected: &[u8], +) -> Result<(), SnapshotError> { + tokio::time::timeout(std::time::Duration::from_secs(5), async { + let (mut stream, meta) = objects + .inner + .get_stream(key) + .await + .map_err(|_| integrity("admitted chunk map trusted receipt is unavailable"))?; + if meta.size != expected.len() as i64 { + return Err(integrity( + "chunk map trusted receipt has invalid total size", + )); + } + let mut offset = 0; + while let Some(part) = stream.next().await { + let bytes = part.map_err(|_| integrity("chunk map trusted receipt stream failed"))?; + if bytes.len() > expected.len() - offset + || &expected[offset..offset + bytes.len()] != bytes.as_ref() + { + return Err(integrity( + "chunk map trusted receipt bytes disagree with source and descriptor", + )); + } + offset += bytes.len(); + } + if offset != expected.len() { + return Err(integrity("chunk map trusted receipt is truncated")); + } + Ok(()) + }) + .await + .map_err(|_| { + SnapshotError::new( + SnapshotErrorCode::TemporaryUnavailable, + "chunk map receipt read timed out", + ) + })? +} + +fn encode_leaves(leaves: &[ChunkLeaf]) -> Result { + let values: Result, SnapshotError> = leaves.iter().map(|l| Ok(json!({"page_index":l.page_index,"payload":hex::encode(l.encode().map_err(|_| integrity("invalid verified leaf"))?)}))).collect(); + serde_json::to_string(&values?).map_err(db_error) +} + +fn encode_nodes(nodes: &[ChunkMapNode]) -> Result { + serde_json::to_string(&nodes.iter().map(|n| json!({"first_page":n.start,"page_count":n.pages,"digest":hex::encode(n.digest)})).collect::>()).map_err(db_error) +} + +async fn compare_complete_map( + txn: &DatabaseTransaction, + verified: &VerifiedSourceChunkMap, + nodes: &[ChunkMapNode], + schema: &str, +) -> Result<(), SnapshotError> { + let map = verified.map(); + let row = txn.query_one_raw(stmt(schema, "SELECT CASE WHEN pg_catalog.octet_length(descriptor)=100 THEN descriptor ELSE NULL END AS descriptor,page_count,CASE WHEN pg_catalog.octet_length(pages_root)=32 THEN pages_root ELSE NULL END AS pages_root,(SELECT pg_catalog.count(*) FROM mst2_chunk_map_leaf WHERE map_id=$1) AS leaves,(SELECT pg_catalog.count(*) FROM mst2_chunk_map_node WHERE map_id=$1) AS nodes FROM mst2_chunk_map WHERE map_id=$1", [map.map_id().to_vec().into()])) + .await.map_err(db_error)?.ok_or_else(|| integrity("chunk map installation descriptor is missing"))?; + if bounded_bytes(&row, "descriptor")? != map.encode() + || row.try_get::("", "page_count").map_err(db_error)? as u64 != map.page_count + || bounded_bytes(&row, "pages_root")? != map.pages_root + || row.try_get::("", "leaves").map_err(db_error)? as u64 != map.page_count + || row.try_get::("", "nodes").map_err(db_error)? as usize != nodes.len() + { + return Err(integrity( + "chunk map installation has conflicting descriptor or incomplete coverage", + )); + } + for batch in verified.leaves().chunks(64) { + let bad = txn.query_one_raw(stmt(schema, "SELECT p.page_index FROM pg_catalog.jsonb_to_recordset($2::jsonb) AS p(page_index integer,payload text) LEFT JOIN mst2_chunk_map_leaf l ON l.map_id=$1 AND l.page_index=p.page_index WHERE l.payload IS DISTINCT FROM pg_catalog.decode(p.payload,'hex') LIMIT 1", + [map.map_id().to_vec().into(), encode_leaves(batch)?.into()])).await.map_err(db_error)?; + if bad.is_some() { + return Err(integrity( + "immutable stored chunk map leaf conflicts with verified source", + )); + } + } + for batch in nodes.chunks(1024) { + let bad = txn.query_one_raw(stmt(schema, "SELECT p.first_page FROM pg_catalog.jsonb_to_recordset($2::jsonb) AS p(first_page integer,page_count integer,digest text) LEFT JOIN mst2_chunk_map_node n ON n.map_id=$1 AND n.first_page=p.first_page AND n.page_count=p.page_count WHERE n.digest IS DISTINCT FROM pg_catalog.decode(p.digest,'hex') LIMIT 1", + [map.map_id().to_vec().into(), encode_nodes(batch)?.into()])).await.map_err(db_error)?; + if bad.is_some() { + return Err(integrity( + "immutable stored chunk map node conflicts with verified source", + )); + } + } + Ok(()) +} + +fn quoted(schema: &str) -> String { + format!("\"{}\"", schema.replace('"', "\"\"")) +} + +/// The static statements use these exact relation tokens. Catalog functions +/// are explicitly qualified; relations never resolve through pg_temp. +fn stmt(schema: &str, sql: &str, values: [sea_orm::Value; N]) -> Statement { + Statement::from_sql_and_values(DbBackend::Postgres, qualify_relations(schema, sql), values) +} + +fn qualify_relations(schema: &str, sql: &str) -> String { + let mut qualified = String::with_capacity(sql.len() + 256); + let prefix = quoted(schema); + let bytes = sql.as_bytes(); + let mut cursor = 0; + while cursor < bytes.len() { + let start = cursor; + let byte = bytes[cursor]; + if byte == b'\'' || byte == b'"' { + cursor += 1; + while cursor < bytes.len() { + if bytes[cursor] == byte { + cursor += 1; + if cursor < bytes.len() && bytes[cursor] == byte { + cursor += 1; + } else { + break; + } + } else { + cursor += 1; + } + } + } else if byte.is_ascii_alphanumeric() || byte == b'_' { + cursor += 1; + while cursor < bytes.len() + && (bytes[cursor].is_ascii_alphanumeric() || bytes[cursor] == b'_') + { + cursor += 1; + } + if [ + "mst2_metadata_storage_scope", + "mst2_verified_object", + "mst2_chunk_map", + "mst2_chunk_map_source", + "mst2_chunk_map_leaf", + "mst2_chunk_map_node", + ] + .contains(&&sql[start..cursor]) + { + qualified.push_str(&prefix); + qualified.push('.'); + } + } else { + cursor += sql[cursor..].chars().next().map_or(1, char::len_utf8); + } + qualified.push_str(&sql[start..cursor]); + } + qualified +} + +#[cfg(test)] +mod sql_tests { + use super::*; + + #[test] + fn authority_relations_are_qualified_without_rewriting_catalog_literals() { + let sql = "SELECT 'mst2_metadata_storage_scope','quoted ''mst2_chunk_map''',\"mst2_chunk_map\" FROM mst2_metadata_storage_scope s JOIN mst2_verified_object f ON s.singleton=1 JOIN mst2_chunk_map m ON true JOIN mst2_chunk_map_leaf l ON true JOIN mst2_chunk_map_node n ON true JOIN mst2_chunk_map_source r ON true WHERE c.relname='mst2_metadata_storage_scope'"; + let qualified = qualify_relations("schema\"name", sql); + assert!(qualified.contains(&format!( + "FROM {}.mst2_metadata_storage_scope", + quoted("schema\"name") + ))); + for name in [ + "mst2_verified_object", + "mst2_chunk_map", + "mst2_chunk_map_leaf", + "mst2_chunk_map_node", + "mst2_chunk_map_source", + ] { + assert!(qualified.contains(&format!("JOIN {}.{name}", quoted("schema\"name")))); + } + assert!(qualified.starts_with( + "SELECT 'mst2_metadata_storage_scope','quoted ''mst2_chunk_map''',\"mst2_chunk_map\"" + )); + assert!(qualified.ends_with("WHERE c.relname='mst2_metadata_storage_scope'")); + } +} +fn bounded_bytes(row: &QueryResult, name: &str) -> Result, SnapshotError> { + row.try_get::>>("", name) + .map_err(db_error)? + .ok_or_else(|| { + integrity("persisted chunk map field is missing or outside its bounded profile") + }) +} +fn integrity(message: &str) -> SnapshotError { + SnapshotError::new(SnapshotErrorCode::IntegrityError, message) +} +fn db_error(error: impl std::fmt::Display) -> SnapshotError { + tracing::warn!(%error, "persisted chunk map storage operation failed"); + SnapshotError::new( + SnapshotErrorCode::TemporaryUnavailable, + "persisted chunk map storage operation failed", + ) +} +fn storage_error(error: impl std::fmt::Display) -> SnapshotError { + tracing::warn!(%error, "trusted chunk map receipt publication failed"); + SnapshotError::new( + SnapshotErrorCode::TemporaryUnavailable, + "trusted chunk map receipt publication failed", + ) +} +async fn finish( + txn: DatabaseTransaction, + result: Result, +) -> Result { + match result { + Ok(value) => { + txn.commit().await.map_err(db_error)?; + Ok(value) + } + Err(error) => { + let _ = txn.rollback().await; + Err(error) + } + } +} + +impl super::Storage { + pub(crate) async fn chunk_maps(&self) -> Result<&PostgresChunkMapRepository, SnapshotError> { + use super::base_storage::StorageConnector; + self.native_chunk_maps + .get_or_try_init(|| async { + PostgresChunkMapRepository::new(self.mono_storage().get_connection().clone()).await + }) + .await + } +} diff --git a/src/jupiter/storage/object_storage.rs b/src/jupiter/storage/object_storage.rs index 506f2974..f65a33ab 100644 --- a/src/jupiter/storage/object_storage.rs +++ b/src/jupiter/storage/object_storage.rs @@ -78,6 +78,110 @@ mod exact_range_tests { ); } } + + #[tokio::test] + async fn memory_chunk_receipts_reject_all_mutation_routes() { + let storage = mock_object_storage(); + super::assert_immutable_chunk_receipt_contract(storage.inner.as_ref()).await; + } +} + +#[cfg(test)] +pub(crate) async fn assert_immutable_chunk_receipt_contract( + storage: &dyn crate::orbit_api::factory::MegaObjectStorageWithLog, +) { + let key = ObjectKey { + namespace: crate::orbit_api::object_storage::ObjectNamespace::ChunkMapReceipt, + key: "a".repeat(64), + }; + let first = Bytes::from_static(b"trusted immutable receipt"); + storage + .put_metadata_atomic_create(&key, first.clone(), ObjectMeta::default()) + .await + .unwrap(); + storage + .put_metadata_atomic_create( + &key, + Bytes::from_static(b"conflicting replay"), + ObjectMeta::default(), + ) + .await + .unwrap(); + let input = || { + Box::pin(futures::stream::iter([Ok(Bytes::from_static( + b"overwrite", + ))])) as ObjectByteStream + }; + assert!( + storage + .put_stream(&key, input(), ObjectMeta::default()) + .await + .is_err() + ); + assert!( + storage + .put_stream_bounded(&key, input(), ObjectMeta::default()) + .await + .is_err() + ); + assert!( + storage + .put_metadata_atomic( + &key, + Bytes::from_static(b"overwrite"), + ObjectMeta::default() + ) + .await + .is_err() + ); + assert!( + storage + .put_metadata_atomic_create( + &key, + Bytes::from(vec![ + 0; + crate::orbit_api::object_storage::MAX_METADATA_ATOMIC_BYTES + + 1 + ]), + ObjectMeta::default() + ) + .await + .is_err() + ); + assert!( + storage + .append(&key, input(), ObjectMeta::default()) + .await + .is_err() + ); + assert!( + storage + .append_concurrently(&key, input(), ObjectMeta::default()) + .await + .is_err() + ); + assert!(storage.delete(&key).await.is_err()); + for method in [Method::PUT, Method::POST, Method::DELETE, Method::PATCH] { + assert!( + storage + .signed_url(&key, method, std::time::Duration::from_secs(60)) + .await + .is_err() + ); + } + assert!( + storage + .signed_url(&key, Method::GET, std::time::Duration::from_secs(60)) + .await + .is_ok() + ); + let (mut stream, meta) = storage.get_stream(&key).await.unwrap(); + assert_eq!(meta.size, first.len() as i64); + let mut observed = Vec::new(); + while let Some(part) = stream.next().await { + observed.extend_from_slice(&part.unwrap()); + } + assert_eq!(observed, first.as_ref()); } #[derive(Default)] @@ -111,6 +215,7 @@ impl MegaObjectStorage for InMemoryObjectStorage { data: ObjectByteStream, mut meta: ObjectMeta, ) -> OrbitResult<()> { + reject_receipt_mutation(key)?; let bytes = Self::read_stream(data).await?; meta.size = bytes.len() as i64; self.objects @@ -126,6 +231,7 @@ impl MegaObjectStorage for InMemoryObjectStorage { bytes: Bytes, mut meta: ObjectMeta, ) -> OrbitResult<()> { + reject_receipt_mutation(key)?; use crate::orbit_api::object_storage::MAX_METADATA_ATOMIC_BYTES; if bytes.len() > MAX_METADATA_ATOMIC_BYTES { return Err(IoOrbitError::Other(format!( @@ -140,6 +246,27 @@ impl MegaObjectStorage for InMemoryObjectStorage { Ok(()) } + async fn put_metadata_atomic_create( + &self, + key: &ObjectKey, + bytes: Bytes, + mut meta: ObjectMeta, + ) -> OrbitResult<()> { + if bytes.len() > crate::orbit_api::object_storage::MAX_METADATA_ATOMIC_BYTES { + return Err(IoOrbitError::Other( + "immutable atomic metadata exceeds its byte limit".into(), + )); + } + key.validate()?; + meta.size = bytes.len() as i64; + self.objects + .lock() + .map_err(|_| IoOrbitError::Other("object storage lock poisoned".into()))? + .entry(key.clone()) + .or_insert((bytes, meta)); + Ok(()) + } + async fn get_stream(&self, key: &ObjectKey) -> OrbitResult<(ObjectByteStream, ObjectMeta)> { let (bytes, meta) = self .objects @@ -208,14 +335,18 @@ impl MegaObjectStorage for InMemoryObjectStorage { async fn signed_url( &self, - _key: &ObjectKey, - _method: Method, + key: &ObjectKey, + method: Method, _expires_in: std::time::Duration, ) -> OrbitResult> { + if method != Method::GET { + reject_receipt_mutation(key)?; + } Ok(None) } async fn delete(&self, key: &ObjectKey) -> OrbitResult<()> { + reject_receipt_mutation(key)?; let deleted = self .objects .lock() @@ -230,6 +361,16 @@ impl MegaObjectStorage for InMemoryObjectStorage { } } +fn reject_receipt_mutation(key: &ObjectKey) -> OrbitResult<()> { + if key.namespace == crate::orbit_api::object_storage::ObjectNamespace::ChunkMapReceipt { + Err(IoOrbitError::Other( + "chunk map receipts require immutable atomic creation".into(), + )) + } else { + Ok(()) + } +} + #[async_trait::async_trait] impl LogStorage for InMemoryObjectStorage { async fn append( @@ -238,6 +379,7 @@ impl LogStorage for InMemoryObjectStorage { data: ObjectByteStream, mut meta: ObjectMeta, ) -> OrbitResult<()> { + reject_receipt_mutation(key)?; let bytes = Self::read_stream(data).await?; let mut objects = self .objects diff --git a/src/orbit/adapter/log.rs b/src/orbit/adapter/log.rs index 4d116744..6dd1d78f 100644 --- a/src/orbit/adapter/log.rs +++ b/src/orbit/adapter/log.rs @@ -8,6 +8,7 @@ impl LogStorage for ObjectStoreAdapter { data: ObjectByteStream, _meta: ObjectMeta, ) -> OrbitResult<()> { + super::object::reject_receipt_mutation(key)?; key.validate()?; // Fast path for single writer/single thread (optimized here): // - No conditional writes, no retries, no cleanup (no concurrent write conflicts) @@ -187,6 +188,7 @@ impl LogStorage for ObjectStoreAdapter { data: ObjectByteStream, _meta: ObjectMeta, ) -> OrbitResult<()> { + super::object::reject_receipt_mutation(key)?; key.validate()?; // The local backend has no conditional-write (CAS) support, so this // method cannot be made safe under contention there (it would fall back diff --git a/src/orbit/adapter/object.rs b/src/orbit/adapter/object.rs index 58ccacc0..87327a39 100644 --- a/src/orbit/adapter/object.rs +++ b/src/orbit/adapter/object.rs @@ -12,6 +12,7 @@ impl MegaObjectStorage for ObjectStoreAdapter { data: ObjectByteStream, _meta: ObjectMeta, ) -> OrbitResult<()> { + reject_receipt_mutation(key)?; let path = Self::checked_path(key)?; // Artifacts are keyed by UUID and must not depend on LFS/Git upload_strategy // (e.g. S3 often uses `Multipart` for LFS while we still need create-if-absent semantics). @@ -35,6 +36,7 @@ impl MegaObjectStorage for ObjectStoreAdapter { data: ObjectByteStream, _meta: ObjectMeta, ) -> OrbitResult<()> { + reject_receipt_mutation(key)?; let path = Self::checked_path(key)?; self.put_multipart(&path, data).await } @@ -45,6 +47,7 @@ impl MegaObjectStorage for ObjectStoreAdapter { bytes: Bytes, _meta: ObjectMeta, ) -> OrbitResult<()> { + reject_receipt_mutation(key)?; use crate::orbit_api::object_storage::MAX_METADATA_ATOMIC_BYTES; if bytes.len() > MAX_METADATA_ATOMIC_BYTES { return Err(IoOrbitError::Other(format!( @@ -61,6 +64,32 @@ impl MegaObjectStorage for ObjectStoreAdapter { Ok(()) } + async fn put_metadata_atomic_create( + &self, + key: &ObjectKey, + bytes: Bytes, + _meta: ObjectMeta, + ) -> OrbitResult<()> { + if bytes.len() > crate::orbit_api::object_storage::MAX_METADATA_ATOMIC_BYTES { + return Err(IoOrbitError::Other( + "immutable atomic metadata exceeds its byte limit".into(), + )); + } + let path = Self::checked_path(key)?; + match self + .to_store() + .put_opts( + &path, + PutPayload::from_bytes(bytes), + PutOptions::from(PutMode::Create), + ) + .await + { + Ok(_) | Err(object_store::Error::AlreadyExists { .. }) => Ok(()), + Err(error) => Err(IoOrbitError::from(error)), + } + } + async fn get_stream(&self, key: &ObjectKey) -> OrbitResult<(ObjectByteStream, ObjectMeta)> { let path = Self::checked_path(key)?; @@ -149,6 +178,11 @@ impl MegaObjectStorage for ObjectStoreAdapter { method: Method, expires_in: Duration, ) -> OrbitResult> { + if key.namespace == ObjectNamespace::ChunkMapReceipt && method != Method::GET { + return Err(IoOrbitError::Other( + "immutable chunk map receipts do not permit presigned mutation".into(), + )); + } let path = Self::checked_path(key)?; let url = match &self.store { @@ -178,6 +212,7 @@ impl MegaObjectStorage for ObjectStoreAdapter { } async fn delete(&self, key: &ObjectKey) -> OrbitResult<()> { + reject_receipt_mutation(key)?; let path = Self::checked_path(key)?; self.to_store() .delete(&path) @@ -187,10 +222,34 @@ impl MegaObjectStorage for ObjectStoreAdapter { } } +pub(super) fn reject_receipt_mutation(key: &ObjectKey) -> OrbitResult<()> { + if key.namespace == ObjectNamespace::ChunkMapReceipt { + Err(IoOrbitError::Other( + "chunk map receipts require immutable atomic creation".into(), + )) + } else { + Ok(()) + } +} + #[cfg(test)] mod exact_range_tests { use super::*; + #[tokio::test] + async fn local_chunk_receipts_reject_all_mutation_routes() { + let directory = tempfile::tempdir().unwrap(); + let adapter = ObjectStoreAdapter { + store: BackendStore::Local(Arc::new( + LocalFileSystem::new_with_prefix(directory.path()).unwrap(), + )), + upload_strategy: UploadStrategy::SinglePut, + presign_store: None, + }; + crate::jupiter::storage::object_storage::assert_immutable_chunk_receipt_contract(&adapter) + .await; + } + #[tokio::test] async fn local_exact_range_returns_selected_bytes_and_full_object_size_without_clipping() { let directory = tempfile::tempdir().unwrap(); diff --git a/src/orbit_api/object_storage.rs b/src/orbit_api/object_storage.rs index 7fe6f47a..90539315 100644 --- a/src/orbit_api/object_storage.rs +++ b/src/orbit_api/object_storage.rs @@ -122,6 +122,8 @@ pub enum ObjectNamespace { Media, /// Agent Capture objects (`docs/refactoring/agent-capture.md`). Agent, + /// Immutable receipts minted by the full-stream chunk-map verifier. + ChunkMapReceipt, } impl ObjectNamespace { @@ -135,6 +137,7 @@ impl ObjectNamespace { ObjectNamespace::Oci => "oci", ObjectNamespace::Media => "media", ObjectNamespace::Agent => "agent", + ObjectNamespace::ChunkMapReceipt => "chunk-map-receipt", } } } @@ -260,6 +263,21 @@ pub trait MegaObjectStorage: Send + Sync { )) } + /// Create a complete <=1 MiB immutable metadata object. An existing key + /// must retain its original bytes. The caller must read and compare the + /// existing object before treating an idempotent replay as success. + async fn put_metadata_atomic_create( + &self, + _key: &ObjectKey, + _bytes: Bytes, + _meta: ObjectMeta, + ) -> OrbitResult<()> { + Err(IoOrbitError::Other( + "immutable atomic metadata creation is not supported by this storage backend" + .to_string(), + )) + } + /// Retrieve a single object from the storage backend. /// /// # Returns @@ -527,6 +545,7 @@ mod tests { ObjectNamespace::Oci, ObjectNamespace::Media, ObjectNamespace::Agent, + ObjectNamespace::ChunkMapReceipt, ] { let key = ObjectKey { namespace: ns, @@ -550,6 +569,10 @@ mod tests { assert_eq!(ObjectNamespace::Oci.to_string(), "oci"); assert_eq!(ObjectNamespace::Media.to_string(), "media"); assert_eq!(ObjectNamespace::Agent.to_string(), "agent"); + assert_eq!( + ObjectNamespace::ChunkMapReceipt.to_string(), + "chunk-map-receipt" + ); } #[test]